From 2c65c8e3c09baf4a445b30e8dc7c07aab76af144 Mon Sep 17 00:00:00 2001
From: zfarbp <19612069+zfarbp@users.noreply.github.com>
Date: Thu, 24 Sep 2026 18:44:59 +0000
Subject: [PATCH] chore: refresh bundled models.dev prices
---
packages/server/data/models-pricing.json | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/packages/server/data/models-pricing.json b/packages/server/data/models-pricing.json
index 229ee1c..e19f236 100644
--- a/packages/server/data/models-pricing.json
+++ b/packages/server/data/models-pricing.json
@@ -1 +1 @@
-{"subconscious":{"id":"subconscious","env":["SUBCONSCIOUS_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.subconscious.dev/v1","name":"Subconscious","doc":"https://docs.subconscious.dev","models":{"subconscious/glm-5.2":{"id":"subconscious/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"subconscious/tim-qwen3.6-27b":{"id":"subconscious/tim-qwen3.6-27b","name":"TIM-Qwen3.6 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":5000},"cost":{"input":0.3,"output":3,"cache_read":0.15}}}},"tokengo":{"id":"tokengo","env":["TOKENGO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokengo.com/v1","name":"TokenGo","doc":"https://www.tokengo.com/docs","models":{"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":2.65,"cache_read":0.2}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.098,"output":0.196,"cache_read":0.028}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.2174,"output":0.326,"cache_read":0.06}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.19,"output":0.71,"cache_read":0.06}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.025,"cache_read":0.015}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.89,"output":3.2647,"cache_read":0.2226}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"modelis":{"id":"modelis","env":["MODELIS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://modelishub.com/v1","name":"Modelis","doc":"https://modelishub.com/pricing","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0983,"output":0.1966}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]},{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":3,"output":9}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.768,"output":3.072}}}},"bothub":{"id":"bothub","env":["BOTHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.bothub.ru/v1","name":"Bothub","doc":"https://bothub.ru/models","models":{"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.61,"output":4.84}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1,"output":0.28}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.06,"output":0.37}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.44}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.72,"output":5.41}}}},"greenpt":{"id":"greenpt","env":["GREENPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.greenpt.ai/v1","name":"GreenPT","doc":"https://docs.greenpt.ai","models":{"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1596,"output":0.399,"cache_read":0.0456}},"glm-5.2-caveman-ultra":{"id":"glm-5.2-caveman-ultra","name":"GLM-5.2 Caveman Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-ponytail-ultra":{"id":"glm-5.2-ponytail-ultra","name":"GLM-5.2 Ponytail Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-honey-ultra":{"id":"glm-5.2-honey-ultra","name":"GLM-5.2 Honey Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7524,"output":4.275,"cache_read":0.2508}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.1938,"output":1.129,"cache_read":0.0627}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9006,"output":4.389,"cache_read":0.1881}},"green-l":{"id":"green-l","name":"Green L","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"devstral-2-123b-instruct-2512":{"id":"devstral-2-123b-instruct-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":16384},"cost":{"input":0.57,"output":2.736}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.255552,"output":1.27776,"cache_read":0.0127776}},"glm-5.2-honey-lite":{"id":"glm-5.2-honey-lite","name":"GLM-5.2 Honey Lite","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":1.083}},"green-l-raw":{"id":"green-l-raw","name":"Green L Raw","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.798,"output":4.959}},"glm-5.2-caveman":{"id":"glm-5.2-caveman","name":"GLM-5.2 Caveman","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.762,"output":18.81,"cache_read":0.9405}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"Gemma 3 27B","description":"Google Gemma 3 multimodal model for chat, reasoning, and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40000,"output":8192},"cost":{"input":0.342,"output":0.684}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.342,"output":2.052}},"green-s":{"id":"green-s","name":"Green S","description":"GreenPT speech-to-text model for pre-recorded and live transcription","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.127754,"output":0.511016,"cache_read":0.0255508}},"glm-5.2-caveman-lite":{"id":"glm-5.2-caveman-lite","name":"GLM-5.2 Caveman Lite","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":2.052,"output":10.26}},"glm-5.2-ponytail":{"id":"glm-5.2-ponytail","name":"GLM-5.2 Ponytail","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"gemma4":{"id":"gemma4","name":"gemma4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.57,"output":1.71}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.228,"output":0.456}},"glm-5.2-honey":{"id":"glm-5.2-honey","name":"GLM-5.2 Honey","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"status":"deprecated","cost":{"input":1.756,"output":5.518}},"glm-5.2-ponytail-lite":{"id":"glm-5.2-ponytail-lite","name":"GLM-5.2 Ponytail Lite","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen3 235B MoE instruct model for long-context multilingual chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1.026,"output":3.078}},"green-r-raw":{"id":"green-r-raw","name":"Green R Raw","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.228,"output":0.798}},"holo2-30b-a3b":{"id":"holo2-30b-a3b","name":"Holo2 30B A3B","description":"H Company Holo2 vision model for GUI navigation and computer-use agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11","last_updated":"2025-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":22016,"output":16384},"cost":{"input":0.399,"output":0.969}},"voxtral-small-24b-2507":{"id":"voxtral-small-24b-2507","name":"Voxtral Small 24B","description":"Mistral Voxtral audio-understanding model for speech and transcription tasks","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.228,"output":0.513}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.27754,"output":5.11016,"cache_read":0.319385}},"green-s-pro":{"id":"green-s-pro","name":"Green S Pro","description":"GreenPT advanced speech-to-text model with multilingual transcription support","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-02","last_updated":"2025-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"kimi-k2.6-fast":{"id":"kimi-k2.6-fast","name":"Kimi K2.6 Fast","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":1.655,"output":8.778}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.285,"output":0.285}},"green-r":{"id":"green-r","name":"Green R","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":1.254,"output":1.254}}}},"qiniu-ai":{"id":"qiniu-ai","env":["QINIU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qnaigc.com/v1","name":"Qiniu","doc":"https://developer.qiniu.com/aitokenapi","models":{"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":4096}},"kling-v2-6":{"id":"kling-v2-6","name":"Kling-V2 6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":99999999,"output":99999999}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":4096}},"gemini-3.0-pro-image-preview":{"id":"gemini-3.0-pro-image-preview","name":"Gemini 3.0 Pro Image Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"gemini-2.0-flash":{"id":"gemini-2.0-flash","name":"Gemini 2.0 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"claude-3.5-sonnet":{"id":"claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8200}},"doubao-seed-2.0-mini":{"id":"doubao-seed-2.0-mini","name":"Doubao Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen-Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":4096}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"gemini-3.0-pro-preview":{"id":"gemini-3.0-pro-preview","name":"Gemini 3.0 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"doubao-seed-1.6":{"id":"doubao-seed-1.6","name":"Doubao-Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"Gemini 2.0 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen 2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"claude-4.0-opus":{"id":"claude-4.0-opus","name":"Claude 4.0 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-06","last_updated":"2025-09-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":12000}},"doubao-seed-2.0-code":{"id":"doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-22","last_updated":"2026-02-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen2.5-vl-7b-instruct":{"id":"qwen2.5-vl-7b-instruct","name":"Qwen 2.5 VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"claude-4.1-opus":{"id":"claude-4.1-opus","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"qwen3-30b-a3b-thinking-2507":{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen3 30b A3b Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":126000,"output":32000}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen-vl-max-2025-01-25":{"id":"qwen-vl-max-2025-01-25","name":"Qwen VL-MAX-2025-01-25","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"doubao-seed-2.0-pro":{"id":"doubao-seed-2.0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536}},"glm-4.5":{"id":"glm-4.5","name":"GLM 4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}},"doubao-seed-1.6-thinking":{"id":"doubao-seed-1.6-thinking","name":"Doubao-Seed 1.6 Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"doubao-1.5-vision-pro":{"id":"doubao-1.5-vision-pro","name":"Doubao 1.5 Vision Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"doubao-seed-1.6-flash":{"id":"doubao-seed-1.6-flash","name":"Doubao-Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":80000}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen3 30b A3b Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"qwen3-vl-30b-a3b-thinking":{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen3-Vl 30b A3b Thinking","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen 3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235b A22B Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek-V3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"doubao-1.5-thinking-pro":{"id":"doubao-1.5-thinking-pro","name":"Doubao 1.5 Thinking Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"gemini-3.0-flash-preview":{"id":"gemini-3.0-flash-preview","name":"Gemini 3.0 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen-max-2025-01-25":{"id":"qwen-max-2025-01-25","name":"Qwen2.5-Max-2025-01-25","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-14","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":4096}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Doubao Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"claude-4.0-sonnet":{"id":"claude-4.0-sonnet","name":"Claude 4.0 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"deepseek-v3.1":{"id":"deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"stepfun-ai/gelab-zero-4b-preview":{"id":"stepfun-ai/gelab-zero-4b-preview","name":"Stepfun-Ai/Gelab Zero 4b Preview","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096}},"meituan/longcat-flash-lite":{"id":"meituan/longcat-flash-lite","name":"Meituan/Longcat-Flash-Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":320000}},"meituan/longcat-flash-chat":{"id":"meituan/longcat-flash-chat","name":"Meituan/Longcat-Flash-Chat","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-05","last_updated":"2025-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Stepfun/Step-3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"output":4096}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"Xiaomi/Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax/Minimax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"Minimax/Minimax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"Minimax/Minimax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"Minimax/Minimax-M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"deepseek/deepseek-math-v2":{"id":"deepseek/deepseek-math-v2","name":"Deepseek/Deepseek-Math-V2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":160000,"output":160000}},"deepseek/deepseek-v3.2-exp-thinking":{"id":"deepseek/deepseek-v3.2-exp-thinking","name":"DeepSeek/DeepSeek-V3.2-Exp-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.1-terminus-thinking":{"id":"deepseek/deepseek-v3.1-terminus-thinking","name":"DeepSeek/DeepSeek-V3.1-Terminus-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-251201":{"id":"deepseek/deepseek-v3.2-251201","name":"Deepseek/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"x-ai/grok-code-fast-1":{"id":"x-ai/grok-code-fast-1","name":"x-AI/Grok-Code-Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000}},"x-ai/grok-4-fast-reasoning":{"id":"x-ai/grok-4-fast-reasoning","name":"X-Ai/Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-non-reasoning":{"id":"x-ai/grok-4.1-fast-non-reasoning","name":"X-Ai/Grok 4.1 Fast Non Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-reasoning":{"id":"x-ai/grok-4.1-fast-reasoning","name":"X-Ai/Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":20000000,"output":2000000}},"x-ai/grok-4-fast":{"id":"x-ai/grok-4-fast","name":"x-AI/Grok-4-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-20","last_updated":"2025-09-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4-fast-non-reasoning":{"id":"x-ai/grok-4-fast-non-reasoning","name":"X-Ai/Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"x-AI/Grok-4.1-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"OpenAI/GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"OpenAI/GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"Z-Ai/GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"Z-AI/GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/autoglm-phone-9b":{"id":"z-ai/autoglm-phone-9b","name":"Z-Ai/Autoglm Phone 9b","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":12800,"output":4096}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"Z-Ai/GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}}}},"ambient":{"id":"ambient","env":["AMBIENT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ambient.xyz/v1","name":"Ambient","doc":"https://ambient.xyz","models":{"ambient/large":{"id":"ambient/large","name":"Ambient Large","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.19,"output":1.14,"cache_read":0.03,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"cache_write":0}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.2,"output":4.2,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0,"cache_write":0}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.08,"output":0.18,"cache_read":0.016,"cache_write":0}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.028,"cache_write":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.2,"cache_write":0}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.69,"output":3.49,"cache_read":0.14,"cache_write":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}}}},"tempr":{"id":"tempr","env":["TEMPR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.temprhq.io/v1","name":"Tempr","doc":"https://temprhq.io/docs/gateway-reference.html","models":{"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}}}},"agentrouter":{"id":"agentrouter","env":["AGENTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://agentrouter.org/v1","name":"AgentRouter","doc":"https://agentrouter.org/docs/opencode.html","models":{"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}}}},"xiaomi-token-plan-cn":{"id":"xiaomi-token-plan-cn","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-cn.xiaomimimo.com/v1","name":"Xiaomi Token Plan (China)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"nano-gpt":{"id":"nano-gpt","env":["NANO_GPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://nano-gpt.com/api/v1","name":"NanoGPT","doc":"https://docs.nano-gpt.com","models":{"glm-4.1v-thinking-flashx":{"id":"glm-4.1v-thinking-flashx","name":"GLM 4.1V Thinking FlashX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat 2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"gemma-4-31b-it-garnet":{"id":"gemma-4-31b-it-garnet","name":"Garnet","description":"Garnet is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemini-2.0-pro-exp-02-05":{"id":"gemini-2.0-pro-exp-02-05","name":"Gemini 2.0 Pro 0205","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.989,"output":7.956,"cache_read":0.49725}},"Meta-Llama-3-1-8B-Instruct-FP8":{"id":"Meta-Llama-3-1-8B-Instruct-FP8","name":"Llama 3.1 8B (decentralized)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.02,"output":0.03,"cache_read":0.01}},"ernie-5.0-thinking-preview":{"id":"ernie-5.0-thinking-preview","name":"Ernie 5.0 Thinking Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":3.5,"cache_read":0.5}},"mercury-coder-small":{"id":"mercury-coder-small","name":"Mercury Coder Small","description":"Model by Inception AI. A diffusion large language model that runs incredibly quickly (500+ tokens/second) while matching Claude 3.5 Haiku and GPT-4o-mini. 1st in speed on Copilot arena, and matching 2nd in quality.","family":"mercury","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"gemma-4-26b-a4b-it-luminous":{"id":"gemma-4-26b-a4b-it-luminous","name":"Luminous Mirror","description":"Luminous Mirror is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemma-4-26b-a4b-it-shadowsiren":{"id":"gemma-4-26b-a4b-it-shadowsiren","name":"Shadow Siren","description":"Shadow Siren is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"auto-model-premium":{"id":"auto-model-premium","name":"Auto model (Premium)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"mistral-code-latest":{"id":"mistral-code-latest","name":"Mistral Code Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"doubao-seed-1-6-250615":{"id":"doubao-seed-1-6-250615","name":"Doubao Seed 1.6","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.204,"output":0.51,"cache_read":0.102}},"Gemma-4-26B-A4B-MeroMero":{"id":"Gemma-4-26B-A4B-MeroMero","name":"Gemma 4 26B A4B MeroMero","description":"Gemma 4 26B A4B MeroMero is an NVFP4 multimodal mixture-of-experts fine-tune for emotive dialogue, relationship scenes, creative writing, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"claw-low":{"id":"claw-low","name":"Claw Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"doubao-seed-2-0-mini-260215":{"id":"doubao-seed-2-0-mini-260215","name":"Doubao Seed 2.0 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.0493,"output":0.4845,"cache_read":0.02465}},"gemma-4-31b-it-gemsicle":{"id":"gemma-4-31b-it-gemsicle","name":"Gemsicle","description":"Gemsicle is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemini-2.5-flash-preview-09-2025-thinking":{"id":"gemini-2.5-flash-preview-09-2025-thinking","name":"Gemini 2.5 Flash Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemma-4-26b-a4b-it-opusdistill":{"id":"gemma-4-26b-a4b-it-opusdistill","name":"Opus Distill","description":"Opus Distill is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"glm-4-plus-0111":{"id":"glm-4-plus-0111","name":"GLM 4 Plus 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":9.996,"output":9.996,"cache_read":4.998}},"gemini-2.0-pro-reasoner":{"id":"gemini-2.0-pro-reasoner","name":"Gemini 2.0 Pro Reasoner","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-05","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1.292,"output":4.998,"cache_read":0.323}},"Qwen3.5-27B-Queen-Derestricted":{"id":"Qwen3.5-27B-Queen-Derestricted","name":"Qwen3.5 27B Queen Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"Gemma-4-31B-Cognitive-Unshackled":{"id":"Gemma-4-31B-Cognitive-Unshackled","name":"Gemma 4 31B Cognitive Unshackled","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"Gemini 2.5 Pro Preview 0605","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"asi1-mini":{"id":"asi1-mini","name":"ASI1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":1,"cache_read":0.5}},"gemini-2.5-pro-preview-03-25":{"id":"gemini-2.5-pro-preview-03-25","name":"Gemini 2.5 Pro Preview 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"glm-4-air-0111":{"id":"glm-4-air-0111","name":"GLM 4 Air 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-11","last_updated":"2025-01-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":0.1394,"output":0.1394,"cache_read":0.0697}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Cohere Command A (08/2025)","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"phi-4-multimodal-instruct":{"id":"phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.07,"output":0.11,"cache_read":0.035}},"mistral-code-agent-latest":{"id":"mistral-code-agent-latest","name":"Mistral Code Agent Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"Qwen3.5-27B-BlueStar-v3-Derestricted":{"id":"Qwen3.5-27B-BlueStar-v3-Derestricted","name":"Qwen3.5 27B BlueStar v3 Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"input":64000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"GLM-4.6-Derestricted-v5":{"id":"GLM-4.6-Derestricted-v5","name":"GLM 4.6 Derestricted v5","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.4,"output":1.5,"cache_read":0.2}},"doubao-seed-1-6-flash-250615":{"id":"doubao-seed-1-6-flash-250615","name":"Doubao Seed 1.6 Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.0374,"output":0.374,"cache_read":0.0187}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"Gemini 2.5 Flash Lite Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"doubao-1.5-pro-256k":{"id":"doubao-1.5-pro-256k","name":"Doubao 1.5 Pro 256k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.799,"output":1.445,"cache_read":0.3995}},"glm-z1-airx":{"id":"glm-z1-airx","name":"GLM Z1 AirX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"ernie-5.1:thinking":{"id":"ernie-5.1:thinking","name":"ERNIE 5.1 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"universal-summarizer":{"id":"universal-summarizer","name":"Universal Summarizer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":30,"output":30}},"venice-uncensored":{"id":"venice-uncensored","name":"Venice Uncensored","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"venice","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-01","last_updated":"2025-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.4}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"Gemini 2.5 Flash Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.1343,"output":0.3349,"cache_read":0.06715}},"gemma-4-26b-a4b-it-moonlight":{"id":"gemma-4-26b-a4b-it-moonlight","name":"Moonlight Dusk","description":"Moonlight Dusk is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"deepseek-chat-cheaper":{"id":"deepseek-chat-cheaper","name":"DeepSeek V3/Chat Cheaper","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"gemma-4-26b-a4b-it-darksoul":{"id":"gemma-4-26b-a4b-it-darksoul","name":"Dark Soul","description":"Dark Soul is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"Gemini 2.5 Flash Lite Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Gemma-4-31B-Queen":{"id":"Gemma-4-31B-Queen","name":"Gemma 4 31B Queen","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"deepseek-r1-sambanova":{"id":"deepseek-r1-sambanova","name":"DeepSeek R1 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":4.998,"output":6.987,"cache_read":2.499}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Doubao Seed 2.0 Code Preview","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.893,"cache_read":0.391}},"ernie-5.1":{"id":"ernie-5.1","name":"ERNIE 5.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"gemma-4-26b-a4b-it-chimerax":{"id":"gemma-4-26b-a4b-it-chimerax","name":"Chimera X","description":"Chimera X is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":4096},"cost":{"input":0.054,"output":0.2124,"cache_read":0.0336}},"deepseek-chat":{"id":"deepseek-chat","name":"DeepSeek V3/Deepseek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"holo3-35b-a3b":{"id":"holo3-35b-a3b","name":"Holo3-35B-A3B","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"hermes-high":{"id":"hermes-high","name":"Hermes High","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"claw-high":{"id":"claw-high","name":"Claw High","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"holo3-35b-a3b:thinking":{"id":"holo3-35b-a3b:thinking","name":"Holo3-35B-A3B Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"doubao-1.5-vision-pro-32k":{"id":"doubao-1.5-vision-pro-32k","name":"Doubao 1.5 Vision Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.459,"output":1.377,"cache_read":0.2295}},"doubao-seed-2-0-lite-260215":{"id":"doubao-seed-2-0-lite-260215","name":"Doubao Seed 2.0 Lite","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.1462,"output":0.8738,"cache_read":0.0731}},"Gemma-4-26B-A4B-MeroMero:thinking":{"id":"Gemma-4-26B-A4B-MeroMero:thinking","name":"Gemma 4 26B A4B MeroMero Thinking","description":"Gemma 4 26B A4B MeroMero with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemini-2.5-flash-preview-04-17:thinking":{"id":"gemini-2.5-flash-preview-04-17:thinking","name":"Gemini 2.5 Flash Preview Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"claw-medium":{"id":"claw-medium","name":"Claw Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"glm-4-long":{"id":"glm-4-long","name":"GLM-4 Long","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":4096},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"Gemma-4-31B-GarnetV2":{"id":"Gemma-4-31B-GarnetV2","name":"Gemma 4 31B Garnet V2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemini-2.5-flash-preview-04-17":{"id":"gemini-2.5-flash-preview-04-17","name":"Gemini 2.5 Flash Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled":{"id":"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled","name":"Gemma 4 31B Claude 4.6 Opus Reasoning Distilled","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"claude","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.0306}},"fastgpt":{"id":"fastgpt","name":"Web Answer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":7.5,"output":7.5}},"gemini-2.5-flash-nothinking":{"id":"gemini-2.5-flash-nothinking","name":"Gemini 2.5 Flash (No Thinking)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"deepseek-reasoner-cheaper":{"id":"deepseek-reasoner-cheaper","name":"Deepseek R1 Cheaper","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"gemini-2.5-flash-lite-preview-09-2025-thinking":{"id":"gemini-2.5-flash-lite-preview-09-2025-thinking","name":"Gemini 2.5 Flash Lite Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"auto-model-standard":{"id":"auto-model-standard","name":"Auto model (Standard)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.1394,"output":1.3328,"cache_read":0.0697}},"Gemma-4-31B-DarkIdol":{"id":"Gemma-4-31B-DarkIdol","name":"Gemma 4 31B DarkIdol","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"auto-model":{"id":"auto-model","name":"Auto model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":0,"output":0}},"gemma-4-31b-it-darkidol":{"id":"gemma-4-31b-it-darkidol","name":"DarkIdol","description":"DarkIdol is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"glm-4.1v-thinking-flash":{"id":"glm-4.1v-thinking-flash","name":"GLM 4.1V Thinking Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"gemma-4-31b-it-fabled":{"id":"gemma-4-31b-it-fabled","name":"Fabled","description":"Fabled is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"agnes-3.0-flash":{"id":"agnes-3.0-flash","name":"Agnes 3.0 Flash","description":"Agnes 3.0 Flash is a low-cost model for coding, tool use, and multi-turn agent tasks. It supports text and image input, optional thinking, and a 512K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.005}},"qwen3-vl-235b-a22b-instruct-original":{"id":"qwen3-vl-235b-a22b-instruct-original","name":"Qwen3 VL 235B A22B Instruct Original","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"Gemini 2.5 Flash 0520","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"gemma-4-31b-it-gembrain":{"id":"gemma-4-31b-it-gembrain","name":"Gembrain","description":"Gembrain is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"longcat-2.0:thinking":{"id":"longcat-2.0:thinking","name":"LongCat 2.0 Thinking","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"celeris-1":{"id":"celeris-1","name":"Celeris 1","description":"Celeris 1 is a diffusion language model built for ultra-low-latency classification, extraction, judging, query rewriting, and other short structured responses.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-07-25","last_updated":"2026-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":2,"output":6,"cache_read":1}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek Chat 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.77,"cache_read":0.135}},"gemma-4-31b-it-novelist":{"id":"gemma-4-31b-it-novelist","name":"Novelist","description":"Novelist is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemma-4-12b-it":{"id":"gemma-4-12b-it","name":"Gemma 4 12B Instruct","description":"Compact Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Doubao Seed 2.0 Pro","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.876,"cache_read":0.391}},"Gemma-4-31B-MeroMero-v2:thinking":{"id":"Gemma-4-31B-MeroMero-v2:thinking","name":"Gemma 4 31B MeroMero v2 Thinking","description":"Gemma 4 31B MeroMero v2 with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"kimi-k2-instruct-fast":{"id":"kimi-k2-instruct-fast","name":"Kimi K2 0711 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-15","last_updated":"2025-07-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"gemma-4-12b-it-semancer":{"id":"gemma-4-12b-it-semancer","name":"Gemma 4 12B Semancer","description":"Gemma 4 12B Semancer is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"Gemma-4-31B-MeroMero-v2":{"id":"Gemma-4-31B-MeroMero-v2","name":"Gemma 4 31B MeroMero v2","description":"Gemma 4 31B MeroMero v2 is a LoRA finetune for emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"auto-model-basic":{"id":"auto-model-basic","name":"Auto model (Basic)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":0.375}},"gemini-2.5-flash-preview-05-20:thinking":{"id":"gemini-2.5-flash-preview-05-20:thinking","name":"Gemini 2.5 Flash 0520 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"gemini-2.5-pro-exp-03-25":{"id":"gemini-2.5-pro-exp-03-25","name":"Gemini 2.5 Pro Experimental 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"phi-4-mini-instruct":{"id":"phi-4-mini-instruct","name":"Phi 4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"hermes-low":{"id":"hermes-low","name":"Hermes Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"nano-gpt-help":{"id":"nano-gpt-help","name":"NanoGPT Help","description":"Text-only NanoGPT support assistant. Questions are processed by the Help inference provider; do not paste secrets or account credentials. Covers the website, models, API, pricing, memory, media generation, and support.","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6000,"input":6000,"output":512},"cost":{"input":0,"output":0}},"gemma-4-12b-it-station-keeper":{"id":"gemma-4-12b-it-station-keeper","name":"Gemma 4 12B StationKeeper","description":"Gemma 4 12B StationKeeper is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-2.5-pro-preview-05-06":{"id":"gemini-2.5-pro-preview-05-06","name":"Gemini 2.5 Pro Preview 0506","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-06","last_updated":"2025-05-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"pokee-isaac":{"id":"pokee-isaac","name":"Pokee-Isaac 28B","description":"Pokee-Isaac is a 28B agentic model with a roughly 10-million-token context window, function calling, and OpenAI-compatible structured output. Pokee bills in $0.01 increments, rounding each non-zero request up to the next cent.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":60000},"cost":{"input":0.15,"output":1,"cache_read":0.075}},"ernie-x1.1-preview":{"id":"ernie-x1.1-preview","name":"ERNIE X1.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"gemini-exp-1206":{"id":"gemini-exp-1206","name":"Gemini 2.0 Pro 1206","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.258,"output":4.998,"cache_read":0.629}},"gemma-4-31b-it-isometry":{"id":"gemma-4-31b-it-isometry","name":"Isometry","description":"Isometry is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemma-4-26b-a4b-it-musica":{"id":"gemma-4-26b-a4b-it-musica","name":"Musica","description":"Musica is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"hermes-medium":{"id":"hermes-medium","name":"Hermes Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"qvq-max":{"id":"qvq-max","name":"Qwen: QvQ Max","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-28","last_updated":"2025-03-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":1.2,"output":4.8,"cache_read":0.6}},"LatitudeGames/Wayfarer-Large-70B-Llama-3.3":{"id":"LatitudeGames/Wayfarer-Large-70B-Llama-3.3","name":"Llama 3.3 70B Wayfarer","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.5}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B (Instruct)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"qwen/qwen3.5-122b-a10b:thinking":{"id":"qwen/qwen3.5-122b-a10b:thinking","name":"Qwen3.5 122B A10B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen 3 14b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.08,"output":0.24,"cache_read":0.04}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen 2.5 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":1.5997,"output":6.392,"cache_read":0.79985}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen 3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_read":0.0325,"cache_write":0.40625}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"qwen/qwen3.8-27b-obliterated":{"id":"qwen/qwen3.8-27b-obliterated","name":"Qwen 3.8 27B Obliterated","description":"Qwen 3.8 27B Obliterated is an open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.2}},"qwen/qwen3.8-27b-queen":{"id":"qwen/qwen3.8-27b-queen","name":"Qwen 3.8 27B Queen","description":"Qwen 3.8 27B Queen is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 262,144-token context window.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen-turbo":{"id":"qwen/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":8192},"cost":{"input":0.04998,"output":0.2006,"cache_read":0.02499}},"qwen/qwen3.7-flash:thinking":{"id":"qwen/qwen3.7-flash:thinking","name":"Qwen3.7 Flash Thinking","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3.5-omni-plus":{"id":"qwen/qwen3.5-omni-plus","name":"Qwen3.5 Omni Plus","description":"Qwen3.5 Omni Plus is Qwen's stronger general multimodal model. We verified live support for text prompts, images, audio files, and direct video URLs on Alibaba's chat-completions-compatible API. Alibaba describes Plus as a comprehensive evolution of Qwen3 Omni with support for over 10 hours of audio input.","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen 3 32b","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/qwen3.6-27b:thinking":{"id":"qwen/qwen3.6-27b:thinking","name":"Qwen3.6 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"qwen/qwen3.5-flash:thinking":{"id":"qwen/qwen3.5-flash:thinking","name":"Qwen3.5 Flash Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"input":262000,"output":65536},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.2,"output":1.5,"cache_read":0.1}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3.8-max:thinking":{"id":"qwen/qwen3.8-max:thinking","name":"Qwen3.8 Max Thinking","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/qwen3.7-max:thinking":{"id":"qwen/qwen3.7-max:thinking","name":"Qwen3.7 Max Thinking","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"qwen/qwen3.5-27b:thinking":{"id":"qwen/qwen3.5-27b:thinking","name":"Qwen3.5 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3.8-27b-fable":{"id":"qwen/qwen3.8-27b-fable","name":"Qwen 3.8 27B Fable","description":"Qwen 3.8 27B Fable is an open-weight multimodal creative finetune for expressive dialogue, long-form storytelling, character work, and roleplay.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"qwen/qwen3.5-omni-flash":{"id":"qwen/qwen3.5-omni-flash","name":"Qwen3.5 Omni Flash","description":"Qwen3.5 Omni Flash is Qwen's fast multimodal model. We verified live support for text prompts, images, audio files, and direct video URLs on Alibaba's chat-completions-compatible API. Alibaba describes Flash as a fully evolved version of Qwen3 Omni with audio input support across 60+ languages.","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":49152,"input":49152,"output":16384}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/qwen3.5-35b-a3b:thinking":{"id":"qwen/qwen3.5-35b-a3b:thinking","name":"Qwen3.5 35B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.17,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"qwen/qwen3.5-plus:thinking":{"id":"qwen/qwen3.5-plus:thinking","name":"Qwen3.5 Plus Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3.8-27b:thinking":{"id":"qwen/qwen3.8-27b:thinking","name":"Qwen3.8 27B Thinking","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":995904,"input":995904,"output":32768},"cost":{"input":0.3995,"output":1.2002,"cache_read":0.19975}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"qwen/qwen3.8-27b-uncensored":{"id":"qwen/qwen3.8-27b-uncensored","name":"Qwen 3.8 27B Uncensored","description":"Qwen 3.8 27B Uncensored is an NVFP4 open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.15,"output":1.2,"cache_read":0.125}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. Significant improvements in general capabilities, including instruction following, logical reasoning, text comprehension, mathematics, science, coding and tool usage.","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"qwen/qwen3.7-plus:thinking":{"id":"qwen/qwen3.7-plus:thinking","name":"Qwen3.7 Plus Thinking","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.19,"output":1.16,"cache_read":0.02,"cache_write":0.24}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":0.14,"output":0.42,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245760,"input":245760,"output":65536},"cost":{"input":1.04,"output":6.24,"cache_read":0.52}},"qwen/qwen3.8-27b-hemmingway":{"id":"qwen/qwen3.8-27b-hemmingway","name":"Qwen 3.8 27B Hemingway","description":"Qwen 3.8 27B Hemingway is an open-weight NVFP4 multimodal creative finetune for long-form prose, character dialogue, storytelling, and roleplay.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen 3 8B","description":"Qwen 3 8B is a 8B model. Supports switching between thinking and non thinking: trigger thinking with /think and /no_think anywhere in a prompt or system message to toggle chain-of-thought reasoning.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.47,"output":0.47,"cache_read":0.235}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen 3 235b A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"qwen/qwen3.5-397b-a17b:thinking":{"id":"qwen/qwen3.5-397b-a17b:thinking","name":"Qwen3.5 397B A17B Thinking","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":6,"cache_read":0.25}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen 3 235b A22B 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/qwen3.6-35b-a3b:thinking":{"id":"qwen/qwen3.6-35b-a3b:thinking","name":"Qwen3.6 35B A3B Thinking","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/qwen3.8-27b-cybersecurity":{"id":"qwen/qwen3.8-27b-cybersecurity","name":"Qwen 3.8 27B Cybersecurity","description":"Qwen 3.8 27B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.6,"cache_read":0.05}},"qwen/qwen-long":{"id":"qwen/qwen-long","name":"Qwen Long 10M","description":"Alibaba's huge context window model. Takes in up to 10 million tokens, which is equivalent to dozens of books.","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-08-01","last_updated":"2024-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":8192},"cost":{"input":0.1003,"output":0.408,"cache_read":0.05015}},"qwen/qwen3-max-2026-01-23":{"id":"qwen/qwen3-max-2026-01-23","name":"Qwen3 Max 2026-01-23","description":"Qwen3 Max is Alibaba's flagship Qwen 3 reasoning model with native tool use (web search, web extractor, code interpreter) and a 256K context window.","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-01-26","last_updated":"2026-01-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"qwen/qwen3.8-27b-uncensored:thinking":{"id":"qwen/qwen3.8-27b-uncensored:thinking","name":"Qwen 3.8 27B Uncensored Thinking","description":"Qwen 3.8 27B Uncensored with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.15,"output":1.2,"cache_read":0.125}},"qwen/qwen2.5-coder-32b-instruct":{"id":"qwen/qwen2.5-coder-32b-instruct","name":"Qwen 2.5 Coder 32b","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3.8-27b-obliterated:thinking":{"id":"qwen/qwen3.8-27b-obliterated:thinking","name":"Qwen 3.8 27B Obliterated Thinking","description":"Qwen 3.8 27B Obliterated with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.2}},"MarinaraSpaghetti/NemoMix-Unleashed-12B":{"id":"MarinaraSpaghetti/NemoMix-Unleashed-12B","name":"NemoMix 12B Unleashed","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Llama 3.1 8b (uncensored)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.8,"output":1.6,"cache_read":0.4}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion 3.0","description":"Aion 3.0 is a GLM-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion 3.0 Mini","description":"Aion 3.0 Mini is a DeepSeek-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16":{"id":"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16","name":"Llama 3.1 70B Celeste v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"ornith-ai/ornith-1.5-35b-a3b":{"id":"ornith-ai/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, tool use, image understanding, and long-context work. This variant disables thinking for faster direct responses.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"ornith-ai/ornith-1.5-35b-a3b:thinking":{"id":"ornith-ai/ornith-1.5-35b-a3b:thinking","name":"Ornith 1.5 35B Thinking","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, reasoning, tool use, image understanding, and long-context work. This variant enables thinking by default.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"abliteration-ai/abliterated-model-large":{"id":"abliteration-ai/abliterated-model-large","name":"Abliterated Model Large","description":"Abliteration.ai's large text reasoning model is derived from GLM-5.2 and supports native tool calling, structured output, automatic prompt caching, and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliteration-ai/abliterated-model-large-v2":{"id":"abliteration-ai/abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"Abliteration.ai's default large text reasoning model is derived from GLM-5.3 for harder reasoning and evaluation workloads, with automatic prompt caching and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliteration-ai/abliterated-model":{"id":"abliteration-ai/abliterated-model","name":"Abliterated Model","description":"Abliteration.ai's multimodal reasoning model supports text and image input, structured output, automatic prompt caching, and a 262K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":262134},"cost":{"input":3,"output":3,"cache_read":0.3}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"input":6144,"output":4096},"cost":{"input":0.799,"output":1.207,"cache_read":0.3995}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"stepfun-ai/step-3.5-flash-2603":{"id":"stepfun-ai/step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"Ternary Bonsai 2 27B","description":"Ternary Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.5,"cache_read":0.0375}},"pamanseau/OpenReasoning-Nemotron-32B":{"id":"pamanseau/OpenReasoning-Nemotron-32B","name":"OpenReasoning Nemotron 32B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"deepseek-ai/DeepSeek-V3.1:thinking":{"id":"deepseek-ai/DeepSeek-V3.1:thinking","name":"DeepSeek V3.1 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/deepseek-v3.2-exp-thinking":{"id":"deepseek-ai/deepseek-v3.2-exp-thinking","name":"DeepSeek V3.2 Exp Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.1-Terminus:thinking":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus:thinking","name":"DeepSeek V3.1 Terminus (Thinking)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":32768},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"deepseek-ai/deepseek-v3.2-exp":{"id":"deepseek-ai/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"VongolaChouko/Starcannon-Unleashed-12B-v1.0":{"id":"VongolaChouko/Starcannon-Unleashed-12B-v1.0","name":"Mistral Nemo Starcannon 12b v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"poolside/laguna-s-2.1:thinking":{"id":"poolside/laguna-s-2.1:thinking","name":"Laguna S 2.1 Thinking","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"featherless-ai/Qwerky-72B":{"id":"featherless-ai/Qwerky-72B","name":"Qwerky 72B","description":"General-purpose chat model for instruction following, writing, and analysis","family":"qwerky","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"stepfun/step-3.7-flash:thinking":{"id":"stepfun/step-3.7-flash:thinking","name":"Step 3.7 Flash Thinking","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-5-preview":{"id":"stepfun/step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"mlabonne/NeuralDaredevil-8B-abliterated":{"id":"mlabonne/NeuralDaredevil-8B-abliterated","name":"Neural Daredevil 8B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.44,"output":0.44,"cache_read":0.22}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":2.006,"output":6.001,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral Small 3.2 24B (2506)","description":"The latest iteration of Mistral Small, version 3.2 (2506) brings enhanced performance and capabilities. With 24 billion parameters, this model delivers state-of-the-art results across text generation tasks with improved efficiency.","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.4,"cache_read":0.1}},"mistralai/devstral-small-2505":{"id":"mistralai/devstral-small-2505","name":"Mistral Devstral Small 2505","description":"OpenHands+Devstral is 100% local 100% open, and is SOTA for the category on SWE-Bench Verified: 46.8% accuracy.","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.06,"output":0.06,"cache_read":0.03}},"mistralai/devstral-2-123b-instruct-2512":{"id":"mistralai/devstral-2-123b-instruct-2512","name":"Devstral 2 123B","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":26214},"cost":{"input":0.1989,"output":0.595,"cache_read":0.09945}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"mistralai/mistral-small-4-119b-2603:thinking":{"id":"mistralai/mistral-small-4-119b-2603:thinking","name":"Mistral Small 4 119B Thinking","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.05}},"mistralai/mistral-nemo-instruct-2407":{"id":"mistralai/mistral-nemo-instruct-2407","name":"Mistral Nemo","description":"12B parameter model with multilingual support.","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 24B","description":"Mistral Small 24B hosted by IONOS in Berlin, Germany. Zero data retention.","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.1155,"output":0.3465}},"mistralai/mixtral-8x22b-instruct-v0.1":{"id":"mistralai/mixtral-8x22b-instruct-v0.1","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B (2503)","description":"Building upon Mistral Small 3 (2501), Mistral Small 3.1 (2503) adds state-of-the-art vision understanding and enhances long context capabilities up to 128k tokens without compromising text performance. With 24 billion parameters, this model achieves top-tier capabilities in both text and vision tasks.","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"mistralai/mistral-medium-3.5:thinking":{"id":"mistralai/mistral-medium-3.5:thinking","name":"Mistral Medium 3.5 Thinking","description":"Mistral Medium 3.5 with reasoning enabled by default (reasoning_effort=high), for complex coding, agentic, and multi-step reasoning prompts.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"mistralai/mistral-medium-3.5":{"id":"mistralai/mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Mistral Medium 3.5 is a 128B dense open-weights flagship model for instruction-following, reasoning, coding, long-horizon agentic work, tool use, structured output, and multimodal prompts. It supports a 256k context window and configurable reasoning effort.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0":{"id":"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0","name":"Omega Directive 24B Unslop v2.0","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated","name":"DeepSeek R1 Llama 70B Abliterated","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated","name":"DeepSeek R1 Qwen Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1.4,"output":1.4,"cache_read":0.7}},"huihui-ai/Llama-3.3-70B-Instruct-abliterated":{"id":"huihui-ai/Llama-3.3-70B-Instruct-abliterated","name":"Llama 3.3 70B Instruct abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/Qwen2.5-32B-Instruct-abliterated":{"id":"huihui-ai/Qwen2.5-32B-Instruct-abliterated","name":"Qwen 2.5 32B Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-06","last_updated":"2025-01-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"xiaomi/mimo-v2.5:thinking":{"id":"xiaomi/mimo-v2.5:thinking","name":"MiMo V2.5 Thinking","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.5-pro:thinking":{"id":"xiaomi/mimo-v2.5-pro:thinking","name":"MiMo V2.5 Pro Thinking","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"shisa-ai/shisa-v2-llama3.3-70b":{"id":"shisa-ai/shisa-v2-llama3.3-70b","name":"Shisa V2 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"shisa-ai/shisa-v2.1-llama3.3-70b":{"id":"shisa-ai/shisa-v2.1-llama3.3-70b","name":"Shisa V2.1 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"minimax/minimax-m3:thinking":{"id":"minimax/minimax-m3:thinking","name":"MiniMax M3 Thinking","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.165}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.17,"output":1.53,"cache_read":0.085}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.315,"output":1.26,"cache_read":0.1575}},"minimax/minimax-m2.7-turbo":{"id":"minimax/minimax-m2.7-turbo","name":"MiniMax M2.7 Turbo","description":"Efficient MiniMax model for quick assistance, coding, and routine automation","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.3}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax M2-her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65532,"input":65532,"output":2048},"cost":{"input":0.302,"output":1.207,"cache_read":0.151}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax 01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"input":1000192,"output":16384},"cost":{"input":0.1394,"output":1.122,"cache_read":0.0697}},"minimax/minimax-latest":{"id":"minimax/minimax-latest","name":"MiniMax Latest","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"Sao10K/L3-8B-Stheno-v3.2":{"id":"Sao10K/L3-8B-Stheno-v3.2","name":"Sao10K Stheno 8b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"Sao10K/L3.3-70B-Euryale-v2.3":{"id":"Sao10K/L3.3-70B-Euryale-v2.3","name":"Llama 3.3 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Sao10K/L3.1-70B-Euryale-v2.2":{"id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Llama 3.1 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.306,"output":0.357,"cache_read":0.153}},"Sao10K/L3.1-70B-Hanami-x1":{"id":"Sao10K/L3.1-70B-Hanami-x1","name":"Llama 3.1 70B Hanami","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"nvidia/nemotron-3-ultra-550b-a55b:thinking":{"id":"nvidia/nemotron-3-ultra-550b-a55b:thinking","name":"Nvidia Nemotron 3 Ultra 550B Thinking","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nvidia Nemotron 3.5 Lightning","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-3-super-120b-a12b:thinking":{"id":"nvidia/nemotron-3-super-120b-a12b:thinking","name":"Nvidia Nemotron 3 Super 120B Thinking","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/nemotron-3.5-lightning:thinking":{"id":"nvidia/nemotron-3.5-lightning:thinking","name":"Nvidia Nemotron 3.5 Lightning Thinking","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nvidia Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF":{"id":"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF","name":"Nvidia Nemotron 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nvidia Nemotron 3 Ultra 550B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1","name":"Nvidia Nemotron Super 49B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nvidia Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"anthropic/claude-opus-4.1:thinking:8192":{"id":"anthropic/claude-opus-4.1:thinking:8192","name":"Claude 4.1 Opus Thinking (8K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4:thinking:8192":{"id":"anthropic/claude-opus-4:thinking:8192","name":"Claude 4 Opus Thinking (8K)","description":"Claude 4 Opus with reduced thinking budget (8,192 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.1:thinking:32768":{"id":"anthropic/claude-opus-4.1:thinking:32768","name":"Claude 4.1 Opus Thinking (32K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4:thinking:8192":{"id":"anthropic/claude-sonnet-4:thinking:8192","name":"Claude 4 Sonnet Thinking (8K)","description":"Claude 4 Sonnet with reduced thinking budget (8,192 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6:thinking":{"id":"anthropic/claude-opus-4.6:thinking","name":"Claude 4.6 Opus Thinking","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4:thinking:1024":{"id":"anthropic/claude-sonnet-4:thinking:1024","name":"Claude 4 Sonnet Thinking (1K)","description":"Claude 4 Sonnet with minimal thinking budget (1,024 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Compatibility alias for Claude Fable.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4:thinking:1024":{"id":"anthropic/claude-opus-4:thinking:1024","name":"Claude 4 Opus Thinking (1K)","description":"Claude 4 Opus with minimal thinking budget (1,024 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.1:thinking:1024":{"id":"anthropic/claude-opus-4.1:thinking:1024","name":"Claude 4.1 Opus Thinking (1K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude 4.7 Opus","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-haiku-4.5:thinking":{"id":"anthropic/claude-haiku-4.5:thinking","name":"Claude Haiku 4.5 Thinking","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4:thinking":{"id":"anthropic/claude-opus-4:thinking","name":"Claude 4 Opus Thinking","description":"Anthropic's Claude 4 Opus with the ability to show its thinking process step by step.","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.6:thinking:low":{"id":"anthropic/claude-opus-4.6:thinking:low","name":"Claude 4.6 Opus Thinking Low","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.1:thinking":{"id":"anthropic/claude-opus-4.1:thinking","name":"Claude 4.1 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6:thinking:medium":{"id":"anthropic/claude-opus-4.6:thinking:medium","name":"Claude 4.6 Opus Thinking Medium","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-5:thinking":{"id":"anthropic/claude-sonnet-5:thinking","name":"Claude Sonnet 5 Thinking","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude 4.6 Opus","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-latest":{"id":"anthropic/claude-haiku-latest","name":"Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-sonnet-4:thinking":{"id":"anthropic/claude-sonnet-4:thinking","name":"Claude 4 Sonnet Thinking","description":"Anthropic's Claude 4 Sonnet with the ability to show its thinking process step by step.","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-4:thinking:32768":{"id":"anthropic/claude-sonnet-4:thinking:32768","name":"Claude 4 Sonnet Thinking (32K)","description":"Claude 4 Sonnet with extended thinking budget (32,768 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-4.6:thinking":{"id":"anthropic/claude-sonnet-4.6:thinking","name":"Claude Sonnet 4.6 Thinking","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude 4 Opus","description":"Claude 4 Opus by Anthropic. The premium version of the new Claude models. A new generation model with improved capabilities, especially on programming and development.","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4:thinking:32768":{"id":"anthropic/claude-opus-4:thinking:32768","name":"Claude 4 Opus Thinking (32K)","description":"Claude 4 Opus with extended thinking budget (32,768 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-4:thinking:64000":{"id":"anthropic/claude-sonnet-4:thinking:64000","name":"Claude 4 Sonnet Thinking (64K)","description":"Claude 4 Sonnet with maximum thinking budget (64,000 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.8:thinking":{"id":"anthropic/claude-opus-4.8:thinking","name":"Claude Opus 4.8 Thinking","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.5:thinking":{"id":"anthropic/claude-opus-4.5:thinking","name":"Claude 4.5 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4.5:thinking":{"id":"anthropic/claude-sonnet-4.5:thinking","name":"Claude Sonnet 4.5 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6:thinking:max":{"id":"anthropic/claude-opus-4.6:thinking:max","name":"Claude 4.6 Opus Thinking Max","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7:thinking":{"id":"anthropic/claude-opus-4.7:thinking","name":"Claude 4.7 Opus Thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude 4 Sonnet","description":"Claude 4 Sonnet by Anthropic. A new generation model with improved capabilities, especially on programming and development. NOTE: Inputs > 200k tokens are charged at 2x input, 1.5x output rate.","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.5-flash-thinking":{"id":"google/gemini-3.5-flash-thinking","name":"Gemini 3.5 Flash Thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-3-flash-preview-thinking":{"id":"google/gemini-3-flash-preview-thinking","name":"Gemini 3 Flash Thinking","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro (Preview Custom Tools)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemma4-31b-splituntied":{"id":"google/gemma4-31b-splituntied","name":"Gemma 4 31B Split-Untied","description":"Blazed-Forge's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use.","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"google/gemini-3.1-pro-preview-high":{"id":"google/gemini-3.1-pro-preview-high","name":"Gemini 3.1 Pro (Preview High)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/diffusiongemma":{"id":"google/diffusiongemma","name":"DiffusionGemma","description":"DiffusionGemma is a high-speed diffusion-based version of Gemma 4 26B A4B. It supports optional reasoning and a 262,144-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","xhigh"]}],"tool_call":false,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemma-4-26b-a4b-it-cybersecurity":{"id":"google/gemma-4-26b-a4b-it-cybersecurity","name":"Gemma 4 26B A4B Cybersecurity","description":"Gemma 4 26B A4B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1056,"output":0.3344,"cache_read":0.0528}},"google/gemma-4-31b-it:thinking":{"id":"google/gemma-4-31b-it:thinking","name":"Gemma 4 31B Thinking","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.35,"cache_read":0.05}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"google/gemma-4-26b-a4b-it:thinking":{"id":"google/gemma-4-26b-a4b-it:thinking","name":"Gemma 4 26B A4B Thinking","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.13,"output":0.4,"cache_read":0.065}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-pro-preview-low":{"id":"google/gemini-3.1-pro-preview-low","name":"Gemini 3.1 Pro (Preview Low)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"baseten/Kimi-K2-Instruct-FP4":{"id":"baseten/Kimi-K2-Instruct-FP4","name":"Kimi K2 0711 Instruct FP4","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling:thinking":{"id":"thinkingmachines/inkling:thinking","name":"Inkling Thinking","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/Inkling-Small:thinking":{"id":"thinkingmachines/Inkling-Small:thinking","name":"Inkling Small Thinking","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Data Used for Training)","description":"A much cheaper opt-in version of Muse Spark 1.2 with the same multimodal coding and agentic capabilities. Prompts and outputs sent to this Contributor model may be used by Meta for training and to improve its products; use the standard Muse Spark 1.2 model if you do not want your data used for training.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Meta's Muse Spark 1.3 Contributor is a frontier multimodal reasoning model for long-horizon coding and agentic workflows, with strong gains in computer use, browsing, professional tool use, codebase understanding, and million-token retrieval. It accepts text, images, audio, video, and files, supports tool calling and structured output, and always reasons before answering. Prompts and outputs may be used by Meta for training and to improve its products.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.15,"output":1.5,"cache_read":0.075}},"Steelskull/L3.3-Cu-Mai-R1-70b":{"id":"Steelskull/L3.3-Cu-Mai-R1-70b","name":"Llama 3.3 70B Cu Mai","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-Electra-R1-70b":{"id":"Steelskull/L3.3-Electra-R1-70b","name":"Steelskull Electra R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.69989,"output":0.69989,"cache_read":0.349945}},"Steelskull/L3.3-Nevoria-R1-70b":{"id":"Steelskull/L3.3-Nevoria-R1-70b","name":"Steelskull Nevoria R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-MS-Nevoria-70b":{"id":"Steelskull/L3.3-MS-Nevoria-70b","name":"Steelskull Nevoria 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"bytedance/doubao-seed-2.1-turbo":{"id":"bytedance/doubao-seed-2.1-turbo","name":"Doubao Seed 2.1 Turbo","description":"Fast, lower-cost model in the Doubao Seed 2.1 family for everyday chat, coding assistance, document work, and high-throughput productivity tasks. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance/doubao-seed-2.1-pro":{"id":"bytedance/doubao-seed-2.1-pro","name":"Doubao Seed 2.1 Pro","description":"Higher-capability model in the Doubao Seed 2.1 family for agentic coding, long-context analysis, complex instruction following, and productivity workflows. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":1,"output":5,"cache_read":0.5}},"bytedance/doubao-seed-character":{"id":"bytedance/doubao-seed-character","name":"Doubao Seed Character","description":"ByteDance's character-focused Doubao Seed model for roleplay, persona consistency, dialogue, and creative character interactions. It supports text and image input with a 128k context window. Requests route through ZenMux to ByteDance; ZenMux does not publish a model-API zero-retention or training guarantee, so avoid sensitive data.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"release_date":"2026-07-18","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1179,"output":0.2947,"cache_read":0.0236,"cache_write":0.0025}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed 2.1 Turbo","description":"ByteDance Seed 2.1 Turbo is a multimodal model for coding and long-horizon agent workflows, including end-to-end software delivery and multi-step task execution. It supports text, image, and video input with a 262k context window.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"ByteDance Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.25}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"ByteDance Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.25,"output":2,"cache_read":0.125}},"NeverSleep/Lumimaid-v0.2-70B":{"id":"NeverSleep/Lumimaid-v0.2-70B","name":"Lumimaid v0.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1,"output":1.5,"cache_read":0.5}},"TEE/qwen3.5-27b":{"id":"TEE/qwen3.5-27b","name":"Qwen3.5 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"TEE/qwen3.8-27b":{"id":"TEE/qwen3.8-27b","name":"Qwen3.8 27B TEE","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"TEE/kimi-k2.6":{"id":"TEE/kimi-k2.6","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.5,"output":5.25,"cache_read":0.375}},"TEE/nemotron-3.5-lightning":{"id":"TEE/nemotron-3.5-lightning","name":"Nvidia Nemotron 3.5 Lightning TEE","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.08,"output":0.2,"cache_read":0.04}},"TEE/glm-5.2":{"id":"TEE/glm-5.2","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/gemma4-31b":{"id":"TEE/gemma4-31b","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-04","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/kimi-k2.7-code":{"id":"TEE/kimi-k2.7-code","name":"Kimi K2.7 Code TEE","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"TEE/qwen2.5-vl-72b-instruct":{"id":"TEE/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"TEE/deepseek-v4.1-flash":{"id":"TEE/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash TEE","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.65,"output":1.45,"cache_read":0.13}},"TEE/gemma4-31b:thinking":{"id":"TEE/gemma4-31b:thinking","name":"Gemma 4 31B Thinking TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-02","last_updated":"2026-05-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/qwen3.5-397b-a17b":{"id":"TEE/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.55,"output":3.5,"cache_read":0.275}},"TEE/gemma-4-26b-a4b-uncensored":{"id":"TEE/gemma-4-26b-a4b-uncensored","name":"Gemma 4 26B A4B Uncensored TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-23","last_updated":"2026-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":65536},"cost":{"input":0.15,"output":0.7,"cache_read":0.075}},"TEE/qwen3.6-27b":{"id":"TEE/qwen3.6-27b","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.32,"output":2.7,"cache_read":0.16}},"TEE/kimi-k3":{"id":"TEE/kimi-k3","name":"Kimi K3 TEE","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":1.5}},"TEE/deepseek-v3.2":{"id":"TEE/deepseek-v3.2","name":"DeepSeek V3.2 TEE","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"input":164000,"output":65536},"cost":{"input":0.5,"output":1,"cache_read":0.25}},"TEE/qwen3.6-35b-a3b":{"id":"TEE/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B TEE","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.2,"output":1.27,"cache_read":0.1}},"TEE/glm-5.3-flash":{"id":"TEE/glm-5.3-flash","name":"GLM 5.3 Flash TEE","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"TEE/muse-glimmer-30b":{"id":"TEE/muse-glimmer-30b","name":"Muse Glimmer 30B TEE","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"TEE/gemma-4-31b-it":{"id":"TEE/gemma-4-31b-it","name":"Gemma 4 31B IT TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.15,"output":0.46,"cache_read":0.075}},"TEE/glm-5.1":{"id":"TEE/glm-5.1","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/glm-5.2:thinking":{"id":"TEE/glm-5.2:thinking","name":"GLM 5.2 Thinking TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/gpt-oss-120b":{"id":"TEE/gpt-oss-120b","name":"GPT-OSS 120B TEE","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":2,"output":2,"cache_read":2}},"TEE/llama3-3-70b":{"id":"TEE/llama3-3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1.75,"output":2.75,"cache_read":1.75}},"TEE/glm-5.1-thinking":{"id":"TEE/glm-5.1-thinking","name":"GLM 5.1 Thinking TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/glm-5.3":{"id":"TEE/glm-5.3","name":"GLM 5.3 TEE","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"meganova-ai/manta-mini-1.0":{"id":"meganova-ai/manta-mini-1.0","name":"Manta Mini 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-flash-1.0":{"id":"meganova-ai/manta-flash-1.0","name":"Manta Flash 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":16384},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-pro-1.0":{"id":"meganova-ai/manta-pro-1.0","name":"Manta Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":0.06,"output":0.5,"cache_read":0.03}},"inception/mercury-2.5-preview":{"id":"inception/mercury-2.5-preview","name":"Mercury 2.5 Preview","description":"Mercury 2.5 Preview is Inception's latest and most intelligent diffusion language model. Instead of generating tokens strictly one at a time, it produces and refines multiple tokens in parallel, reaching up to 1,107 tokens per second on standard GPUs. It delivers a 10+ point intelligence gain over Mercury 2, with tunable reasoning, parallel tool calls, schema-aligned JSON output, and a 260K context window. It is built for latency-sensitive production work such as search agents, voice pipelines, customer support, rapid coding iteration, and coding subagents.","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"input":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"unsloth/gemma-3-4b-it":{"id":"unsloth/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"unsloth/gemma-3-27b-it":{"id":"unsloth/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":96000},"cost":{"input":0.2992,"output":0.2992,"cache_read":0.1496}},"unsloth/gemma-3-12b-it":{"id":"unsloth/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.272,"output":0.272,"cache_read":0.136}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"inflatebot/MN-12B-Mag-Mell-R1":{"id":"inflatebot/MN-12B-Mag-Mell-R1","name":"Mag Mell R1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"deepcogito/cogito-v1-preview-qwen-32B":{"id":"deepcogito/cogito-v1-preview-qwen-32B","name":"Cogito v1 Preview Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-10","last_updated":"2025-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":1.8,"output":1.8,"cache_read":0.9}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Sakana AI's cost-performance Fugu model uses learned multi-agent orchestration to route tasks across expert models for reasoning, coding, and tool use.","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v1.1":{"id":"sakana/fugu-ultra-v1.1","name":"Fugu Ultra v1.1","description":"Sakana AI's upgraded Fugu Ultra release with stronger coding, agentic task execution, and advanced reasoning through dynamic orchestration of frontier models.","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5,"output":30,"cache_read":0.5}},"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B":{"id":"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B","name":"Nemotron Tenyxchat Storybreaker 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B":{"id":"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B","name":"Llama 3.05 Storybreaker Ministral 70b","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"NousResearch/hermes-3-llama-3.1-70b":{"id":"NousResearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-01-07","last_updated":"2026-01-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.408,"output":0.408,"cache_read":0.204}},"NousResearch/hermes-4-405b":{"id":"NousResearch/hermes-4-405b","name":"Hermes 4 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"NousResearch/hermes-4-405b:thinking":{"id":"NousResearch/hermes-4-405b:thinking","name":"Hermes 4 Large (Thinking)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5":{"id":"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5","name":"Llama 3 70B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"GalrionSoftworks/MN-LooseCannon-12B-v1":{"id":"GalrionSoftworks/MN-LooseCannon-12B-v1","name":"MN-LooseCannon-12B-v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"IBM Granite 4.2 8B is an Apache 2.0-licensed dense model with native step-by-step reasoning and specialized training for agentic work. It can plan before acting, sequence tools, navigate codebases, work in terminals, and verify results across coding, search, mathematics, science, and complex instruction-following tasks.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Compatibility alias that routes to the newest dated DeepSeek V4 Flash release. Currently routes to DeepSeek V4 Flash 0731. ⚠️ This route goes directly to DeepSeek, so privacy and logging guarantees are limited.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-v4-flash-0731:thinking":{"id":"deepseek/deepseek-v4-flash-0731:thinking","name":"DeepSeek V4 Flash 0731 (Thinking)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash:thinking":{"id":"deepseek/deepseek-v4-flash:thinking","name":"DeepSeek V4 Flash (Thinking)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.13,"output":0.52,"cache_read":0.006}},"deepseek/deepseek-v4-pro-0813:thinking":{"id":"deepseek/deepseek-v4-pro-0813:thinking","name":"DeepSeek V4 Pro 0813 Thinking","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4.1-flash:thinking":{"id":"deepseek/deepseek-v4.1-flash:thinking","name":"DeepSeek V4.1 Flash Thinking","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.13,"output":0.52,"cache_read":0.006}},"deepseek/deepseek-v4-pro:thinking":{"id":"deepseek/deepseek-v4-pro:thinking","name":"DeepSeek V4 Pro (Thinking)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek/deepseek-latest":{"id":"deepseek/deepseek-latest","name":"DeepSeek Latest","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v3.2:thinking":{"id":"deepseek/deepseek-v3.2:thinking","name":"DeepSeek V3.2 Thinking","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0","name":"EVA Llama 3.33 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1","name":"EVA-LLaMA-3.33-70B-v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2","name":"EVA-Qwen2.5-32B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2","name":"EVA-Qwen2.5-72B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon Nova 2 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65535},"cost":{"input":0.51,"output":4.25,"cache_read":0.255}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":32000},"cost":{"input":0.799,"output":3.196,"cache_read":0.3995}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":5120},"cost":{"input":0.0595,"output":0.238,"cache_read":0.02975}},"LLM360/K2-Think":{"id":"LLM360/K2-Think","name":"K2-Think","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Ling-3.0-flash is a 124B-parameter Mixture-of-Experts model with approximately 5.1B parameters active per token. It prioritizes token efficiency and production-scale agentic inference, helping coding and tool-using agents complete more work within constrained latency and serving budgets.","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash:thinking":{"id":"inclusionai/ling-3.0-flash:thinking","name":"Ling 3.0 Flash Thinking","description":"Ling-3.0-flash Thinking enables visible reasoning on inclusionAI's token-efficient 124B-parameter Mixture-of-Experts model for harder coding, tool use, planning, and production-scale agent workflows.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL is inclusionAI's native multimodal Mixture-of-Experts model with 124B total parameters and 5.5B active parameters per token. It combines image and video understanding with reasoning and tool use for document analysis, charts, visual verification, and interface-based agent tasks. Thinking is enabled by default and can be turned off in settings.","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"MiniMaxAI/MiniMax-M1-80k":{"id":"MiniMaxAI/MiniMax-M1-80k","name":"MiniMax M1 80K","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.6052,"output":2.4225,"cache_read":0.3026}},"lightonai/LightOnOCR-2-1B":{"id":"lightonai/LightOnOCR-2-1B","name":"LightOnOCR 2","description":"LightOnOCR 2 hosted by IONOS in Berlin, Germany. Zero data retention.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.1785,"output":0.3465}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"anthracite-org/magnum-v2-72b":{"id":"anthracite-org/magnum-v2-72b","name":"Magnum V2 72B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"Salesforce/Llama-xLAM-2-70b-fc-r":{"id":"Salesforce/Llama-xLAM-2-70b-fc-r","name":"Llama-xLAM-2 70B fc-r","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":2.5,"cache_read":1.25}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-latest":{"id":"x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8b Instruct","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.0544,"output":0.085,"cache_read":0.0272}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3b Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-09-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.0306,"output":0.0493,"cache_read":0.0153}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":328000,"input":328000,"output":65536},"cost":{"input":0.085,"output":0.46,"cache_read":0.0425}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.05,"output":0.23,"cache_read":0.025}},"abacusai/Dracarys-72B-Instruct":{"id":"abacusai/Dracarys-72B-Instruct","name":"Llama 3.1 70B Dracarys 2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Gryphe/MythoMax-L2-13b":{"id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"input":4096,"output":3686},"cost":{"input":0.1003,"output":0.1003,"cache_read":0.05015}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI o4-mini high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-12-04","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT 5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o3-mini-low":{"id":"openai/o3-mini-low","name":"OpenAI o3-mini (Low)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-01-31","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3-pro-2025-06-10":{"id":"openai/o3-pro-2025-06-10","name":"OpenAI o3-pro (2025-06-10)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":22,"output":88,"cache_read":11}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT 4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT 5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":15,"output":120,"cache_read":1.5}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI o3-mini (High)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT 5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT 6 Astra Pro","description":"GPT 6 Astra in Pro reasoning mode. Uses additional model work for difficult tasks, with higher latency and token usage at the same per-token rates. Reasoning effort remains independently configurable.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT 5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT 5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-2025-11-13":{"id":"openai/gpt-5.1-2025-11-13","name":"GPT-5.1 (2025-11-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT 6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT 5.6 Luna Pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-terra-latest":{"id":"openai/gpt-terra-latest","name":"GPT Terra Latest","description":"Compatibility alias that routes to GPT 5.6 Terra, the latest supported GPT Terra model.","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT 4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT 5.6 Sol Pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.075,"output":0.3}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT 5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/o1":{"id":"openai/o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"OpenAI o1 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":150,"output":600,"cache_read":75}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT 4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT 5.6 Terra Pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-sol-latest":{"id":"openai/gpt-sol-latest","name":"GPT Sol Latest","description":"Compatibility alias that routes to GPT 5.6 Sol, the latest supported GPT Sol model.","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-luna-latest":{"id":"openai/gpt-luna-latest","name":"GPT Luna Latest","description":"Compatibility alias that routes to GPT 5.6 Luna, the latest supported GPT Luna model.","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-astra-latest":{"id":"openai/gpt-astra-latest","name":"GPT Astra Latest","description":"Compatibility alias that routes to GPT 6 Astra, the latest supported GPT Astra model.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT 5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.35,"output":0.75}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT 5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT 5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"OpenAI o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3-mini":{"id":"openai/o3-mini","name":"OpenAI o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"OpenAI o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":1}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"moonshotai/kimi-latest":{"id":"moonshotai/kimi-latest","name":"Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High-Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.9,"output":8,"cache_read":0.32}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":100352},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"moonshotai/kimi-k2.6:thinking":{"id":"moonshotai/kimi-k2.6:thinking","name":"Kimi K2.6 Thinking","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/kimi-k2-instruct-0711":{"id":"moonshotai/kimi-k2-instruct-0711","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.5:thinking":{"id":"moonshotai/kimi-k2.5:thinking","name":"Kimi K2.5 Thinking","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Inference.net's 3B-parameter HTML-to-JSON extraction model, focused on accuracy for complex schemas and long web pages. It turns HTML into typed, structured data for web scraping and product catalog ingestion, with a 128K-token context window. Supply HTML in the user message and extraction instructions in a JSON schema via response_format; it does not follow ordinary chat or system prompts.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.025}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Inference.net's 3B-parameter HTML-to-JSON extraction model, optimized for throughput and low cost on high-volume workloads. It turns HTML into typed, structured data for web scraping and product catalog ingestion, with a 128K-token context window. Supply HTML in the user message and extraction instructions in a JSON schema via response_format; it does not follow ordinary chat or system prompts.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.015}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Cohere: Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":2.856,"output":14.246,"cache_read":1.428}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"upstage/solar-pro4:thinking":{"id":"upstage/solar-pro4:thinking","name":"Solar Pro 4 Thinking","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":80000},"cost":{"input":0.25,"output":0.9,"cache_read":0.125}},"tencent/hy3":{"id":"tencent/hy3","name":"Tencent Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":128000},"cost":{"input":0.066,"output":0.26,"cache_read":0.029}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"TheDrummer/skyfall-36b-v2":{"id":"TheDrummer/skyfall-36b-v2","name":"TheDrummer Skyfall 36B V2","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"TheDrummer/UnslopNemo-12B-v4.1":{"id":"TheDrummer/UnslopNemo-12B-v4.1","name":"UnslopNemo 12b v4","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":26214},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"TheDrummer/Cydonia-24B-v4.3":{"id":"TheDrummer/Cydonia-24B-v4.3","name":"The Drummer Cydonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.12,"output":0.15,"cache_read":0.06}},"TheDrummer/Artemis-v1.1":{"id":"TheDrummer/Artemis-v1.1","name":"TheDrummer/Artemis v1.1","description":"TheDrummer's Artemis v1.1 is a Gemma 4 31B fine-tune for creative writing, expressive dialogue, and roleplay, with optional thinking and a 262K context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-06","last_updated":"2026-09-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"TheDrummer/Cydonia-24B-v2":{"id":"TheDrummer/Cydonia-24B-v2","name":"The Drummer Cydonia 24B v2","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Cydonia-24B-v4":{"id":"TheDrummer/Cydonia-24B-v4","name":"The Drummer Cydonia 24B v4","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.2006,"output":0.2414,"cache_read":0.1003}},"TheDrummer/Anubis-70B-v1.1":{"id":"TheDrummer/Anubis-70B-v1.1","name":"Anubis 70B v1.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Magidonia-24B-v4.3":{"id":"TheDrummer/Magidonia-24B-v4.3","name":"The Drummer Magidonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Cydonia-24B-v4.1":{"id":"TheDrummer/Cydonia-24B-v4.1","name":"The Drummer Cydonia 24B v4.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":0.55,"cache_read":0.16}},"TheDrummer/Anubis-70B-v1":{"id":"TheDrummer/Anubis-70B-v1","name":"Anubis 70B v1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Rocinante-12B-v1.1":{"id":"TheDrummer/Rocinante-12B-v1.1","name":"Rocinante 12b","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.408,"output":0.595,"cache_read":0.204}},"soob3123/Veiled-Calla-12B":{"id":"soob3123/Veiled-Calla-12B","name":"Veiled Calla 12B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/amoral-gemma3-27B-v2":{"id":"soob3123/amoral-gemma3-27B-v2","name":"Amoral Gemma3 27B v2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-23","last_updated":"2025-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/GrayLine-Qwen3-8B":{"id":"soob3123/GrayLine-Qwen3-8B","name":"Grayline Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"nanogpt/coding-router:low":{"id":"nanogpt/coding-router:low","name":"Coding Router Low","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"nanogpt/coding-router:high":{"id":"nanogpt/coding-router:high","name":"Coding Router High","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"nanogpt/coding-router":{"id":"nanogpt/coding-router","name":"Coding Router","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"nanogpt/coding-router:max":{"id":"nanogpt/coding-router:max","name":"Coding Router Max","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"nanogpt/coding-router:medium":{"id":"nanogpt/coding-router:medium","name":"Coding Router Medium","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"liquid/lfm-2.5-2.6b":{"id":"liquid/lfm-2.5-2.6b","name":"LFM2.5 2.6B","description":"Liquid AI's compact 2.6B reasoning model for agent workflows, data extraction, RAG, and long-context processing. It supports tool calling and structured output, but Liquid advises against using it for agentic coding. Warning: prompts and responses may be logged and used for model training or service improvement; do not send sensitive data.","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-4.5v:thinking":{"id":"z-ai/glm-4.5v:thinking","name":"GLM 4.5V Thinking","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"z-ai/glm-4.6v-original":{"id":"z-ai/glm-4.6v-original","name":"GLM 4.6V Original","description":"GLM-4.6V scales its context window to 128k tokens in training, and achieves SoTA performance in visual understanding among models of similar parameter scales. Integrates native Function Calling capabilities, bridging 'visual perception' and 'executable action' for multimodal agents. Direct via Z-AI (Zhipu).","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.6,"output":0.9,"cache_read":0.3}},"z-ai/glm-5.3:thinking":{"id":"z-ai/glm-5.3:thinking","name":"GLM 5.3 Thinking","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/GLM-4.6-turbo":{"id":"z-ai/GLM-4.6-turbo","name":"GLM 4.6 Turbo","description":"Fast variant of GLM 4.6 for general chat, coding, and analysis with improved latency and strong reasoning.","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/GLM-4.5-Air:thinking":{"id":"z-ai/GLM-4.5-Air:thinking","name":"GLM 4.5 Air (Thinking)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-4.6:thinking":{"id":"z-ai/glm-4.6:thinking","name":"GLM 4.6 Thinking","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/glm-4.7-original":{"id":"z-ai/glm-4.7-original","name":"GLM 4.7 Original","description":"GLM-4.7 is a next-gen GLM series text model with stronger reasoning, long-context chat, and reliable tool use. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.7-flash:thinking":{"id":"z-ai/glm-4.7-flash:thinking","name":"GLM 4.7 Flash Thinking","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/glm-4.7:thinking":{"id":"z-ai/glm-4.7:thinking","name":"GLM 4.7 Thinking","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-5.3-flash-cybersecurity":{"id":"z-ai/glm-5.3-flash-cybersecurity","name":"GLM 5.3 Flash Cybersecurity","description":"GLM 5.3 Flash Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports always-on reasoning, image understanding, tool calling, and a 1,048,576-token context window.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":32768},"cost":{"input":0.15,"output":0.5,"cache_read":0.075}},"z-ai/glm-5v-turbo:thinking":{"id":"z-ai/glm-5v-turbo:thinking","name":"GLM 5V Turbo Thinking","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/GLM-4.5:thinking":{"id":"z-ai/GLM-4.5:thinking","name":"GLM 4.5 (Thinking)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/glm-5-original:thinking":{"id":"z-ai/glm-5-original:thinking","name":"GLM 5 Original Thinking","description":"GLM-5 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/GLM-4.5-Air":{"id":"z-ai/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-4.7-original:thinking":{"id":"z-ai/glm-4.7-original:thinking","name":"GLM 4.7 Original Thinking","description":"GLM-4.7 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/glm-5.1:thinking":{"id":"z-ai/glm-5.1:thinking","name":"GLM 5.1 Thinking","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/GLM-4.6-turbo:thinking":{"id":"z-ai/GLM-4.6-turbo:thinking","name":"GLM 4.6 Turbo (Thinking)","description":"GLM 4.6 Turbo with thinking mode enabled for enhanced reasoning; shows internal reasoning and supports long context.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-5.2:thinking":{"id":"z-ai/glm-5.2:thinking","name":"GLM 5.2 Thinking","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/glm-latest":{"id":"z-ai/glm-latest","name":"GLM Latest","description":"Compatibility alias that routes to the newest thinking GLM model. Currently routes to GLM 5.2 Thinking.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5-original":{"id":"z-ai/glm-5-original","name":"GLM 5 Original","description":"GLM-5 is Zhipu's latest flagship model with advanced reasoning and instruction following. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash-uncensored":{"id":"z-ai/glm-5.3-flash-uncensored","name":"GLM 5.3 Flash Uncensored","description":"GLM 5.3 Flash Uncensored is an uncensored fine-tune of the efficient 320B mixture-of-experts reasoning model, built for unrestricted chat, creative writing, coding, agentic work, tool use, and long-context tasks.","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-4.6-original":{"id":"z-ai/glm-4.6-original","name":"GLM 4.6 Original","description":"GLM-4.6, Zhipu's flagship text model with 256K context window and advanced reasoning capabilities. Direct via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-5:thinking":{"id":"z-ai/glm-5:thinking","name":"GLM 5 Thinking","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond":{"id":"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond","name":"MS3.2 24B Magnum Diamond","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"THUDM/GLM-4-9B-0414":{"id":"THUDM/GLM-4-9B-0414","name":"GLM 4 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-Z1-9B-0414":{"id":"THUDM/GLM-Z1-9B-0414","name":"GLM Z1 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-z","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-4-32B-0414":{"id":"THUDM/GLM-4-32B-0414","name":"GLM 4 32B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}}}},"watsonx":{"id":"watsonx","env":["WATSONX_AI_APIKEY","WATSONX_AI_PROJECT_ID"],"npm":"watsonx-ai-provider","name":"watsonx.ai","doc":"https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models","models":{"mistralai/mistral-small-3-1-24b-instruct-2503":{"id":"mistralai/mistral-small-3-1-24b-instruct-2503","name":"Mistral Small 3.1 24B","description":"Efficient multimodal model for instruction following, coding, reasoning, and function calling","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.106,"output":0.318}},"ibm/granite-4-h-small":{"id":"ibm/granite-4-h-small","name":"Granite-4.0-H-Small","description":"Open-weight hybrid model for enterprise chat, coding, retrieval-augmented generation, and tool-calling workloads","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0636,"output":0.265}},"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.371,"output":1.484}},"meta-llama/llama-3-3-70b-instruct":{"id":"meta-llama/llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.7526,"output":0.7526}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.159,"output":0.636}}}},"digitalocean":{"id":"digitalocean","env":["DIGITALOCEAN_ACCESS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.do-ai.run/v1","name":"DigitalOcean","doc":"https://docs.digitalocean.com/products/gradient-ai-platform/details/models/","models":{"openai-gpt-4o":{"id":"openai-gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai-gpt-5.2-pro":{"id":"openai-gpt-5.2-pro","name":"OpenAI GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":21,"output":168}},"bge-reranker-v2-m3":{"id":"bge-reranker-v2-m3","name":"BGE Reranker v2 M3","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-12","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1},"cost":{"input":0.01,"output":0}},"anthropic-claude-opus-4.6":{"id":"anthropic-claude-opus-4.6","name":"Anthropic Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"openai-o3":{"id":"openai-o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.08,"output":0.252,"cache_read":0.0252}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072}},"qwen-2.5-14b-instruct":{"id":"qwen-2.5-14b-instruct","name":"Qwen 2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}},"nvidia-nemotron-3-super-120b":{"id":"nvidia-nemotron-3-super-120b","name":"NVIDIA Nemotron 3 Super 120B (Public Preview)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.3,"output":0.65,"cache_read":0.06}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":1.7,"cache_read":0.09}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"OpenAI GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemma-4-31B-it":{"id":"gemma-4-31B-it","name":"Gemma 4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.18,"output":0.5,"cache_read":0.036}},"alibaba-qwen3-32b":{"id":"alibaba-qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.55}},"openai-gpt-image-1.5":{"id":"openai-gpt-image-1.5","name":"OpenAI GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":5,"output":10,"cache_read":1}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"anthropic-claude-opus-4":{"id":"anthropic-claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"OpenAI GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":2.2,"cache_read":0.105}},"arcee-trinity-large-thinking":{"id":"arcee-trinity-large-thinking","name":"Arcee Trinity Large Thinking (Public Preview)","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.25,"output":0.9,"cache_read":0.06}},"anthropic-claude-opus-4.7":{"id":"anthropic-claude-opus-4.7","name":"Anthropic Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic-claude-fable-5":{"id":"anthropic-claude-fable-5","name":"Anthropic Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic-claude-3.5-sonnet":{"id":"anthropic-claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-06-20","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax M2.5 (Public Preview)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-12","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai-gpt-5.3-codex":{"id":"openai-gpt-5.3-codex","name":"OpenAI GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.055,"output":0.385,"cache_read":0.02}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"OpenAI GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"anthropic-claude-4.5-haiku":{"id":"anthropic-claude-4.5-haiku","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":1,"cache_write":1.25}},"e5-large-v2":{"id":"e5-large-v2","name":"E5 Large v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-05-19","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.02,"output":0}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"OpenAI GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"anthropic-claude-5-sonnet":{"id":"anthropic-claude-5-sonnet","name":"Anthropic Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai-gpt-oss-20b":{"id":"openai-gpt-oss-20b","name":"OpenAI GPT-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.45}},"deepseek-3.2":{"id":"deepseek-3.2","name":"Deepseek 3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":0.8,"cache_read":0.075}},"multi-qa-mpnet-base-dot-v1":{"id":"multi-qa-mpnet-base-dot-v1","name":"Multi-QA-mpnet-base-dot-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":768},"cost":{"input":0.009,"output":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"anthropic-claude-3.7-sonnet":{"id":"anthropic-claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"nemotron-3-nano-30b":{"id":"nemotron-3-nano-30b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gte-large-en-v1.5":{"id":"gte-large-en-v1.5","name":"GTE Large (v1.5)","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-27","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.09,"output":0}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen 3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":3.5,"cache_read":0.111}},"openai-gpt-5":{"id":"openai-gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.25,"output":0.87}},"llama3.3-70b-instruct":{"id":"llama3.3-70b-instruct","name":"Llama 3.3 Instruct (70B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.65,"output":0.65}},"all-mini-lm-l6-v2":{"id":"all-mini-lm-l6-v2","name":"All-MiniLM-L6-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256,"output":384},"cost":{"input":0.009,"output":0}},"anthropic-claude-sonnet-4":{"id":"anthropic-claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.55,"output":12.95,"cache_read":0.285}},"openai-gpt-5.4-mini":{"id":"openai-gpt-5.4-mini","name":"OpenAI GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"anthropic-claude-opus-4.8":{"id":"anthropic-claude-opus-4.8","name":"Anthropic Claude Opus 4.8","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai-gpt-5.4-pro":{"id":"openai-gpt-5.4-pro","name":"OpenAI GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"deepseek-4-flash":{"id":"deepseek-4-flash","name":"Deepseek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-27","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.0679,"output":0.168,"cache_read":0.0168}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"OpenAI GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"tiers":[{"input":8,"output":30,"cache_read":0.8,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8}}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral Nemo Instruct","description":"Legacy model retained for compatibility with older integrations","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.3,"output":0.3}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.9,"output":1.7}},"anthropic-claude-fable-5.1":{"id":"anthropic-claude-fable-5.1","name":"Anthropic Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"nemotron-nano-12b-v2-vl":{"id":"nemotron-nano-12b-v2-vl","name":"Nemotron-nano 12b v2-vl","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.6}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32678,"output":8192},"cost":{"input":0.99,"output":0.99}},"openai-o3-mini":{"id":"openai-o3-mini","name":"OpenAI o3 mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai-gpt-5.1-codex-max":{"id":"openai-gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":0.9}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"OpenAI GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"anthropic-claude-4.5-sonnet":{"id":"anthropic-claude-4.5-sonnet","name":"Anthropic Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-05-22","last_updated":"2024-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768}},"ministral-3-8b-instruct-2512":{"id":"ministral-3-8b-instruct-2512","name":"Ministral 3 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"OpenAI GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"glm-5":{"id":"glm-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"anthropic-claude-3.5-haiku":{"id":"anthropic-claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-11-05","last_updated":"2024-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8-Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.7,"cache_read":0.203}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"mistral-3-14B":{"id":"mistral-3-14B","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.2,"output":0.2}},"qwen3-tts-voicedesign":{"id":"qwen3-tts-voicedesign","name":"Qwen3 TTS VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":32768,"output":1}},"bge-m3":{"id":"bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.02,"output":0}},"qwen3-embedding-0.6b":{"id":"qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":1024},"status":"beta","cost":{"input":0.04,"output":0}},"openai-o1":{"id":"openai-o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"wan2-2-t2v-a14b":{"id":"wan2-2-t2v-a14b","name":"Wan2.2-T2V-A14B","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["video"]},"open_weights":true,"limit":{"context":100,"output":1},"cost":{"input":0.6,"output":0}},"anthropic-claude-3-opus":{"id":"anthropic-claude-3-opus","name":"Claude 3 Opus","description":"Legacy model retained for compatibility with older integrations","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic-claude-4.1-opus":{"id":"anthropic-claude-4.1-opus","name":"Anthropic Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"openai-gpt-image-1":{"id":"openai-gpt-image-1","name":"GPT Image 1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"Deepseek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.87,"output":1.74,"cache_read":0.174}},"stable-diffusion-3.5-large":{"id":"stable-diffusion-3.5-large","name":"Stable Diffusion 3.5 Large","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-10-22","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":256,"output":1},"cost":{"input":0.08,"output":0}},"anthropic-claude-opus-4.5":{"id":"anthropic-claude-opus-4.5","name":"Anthropic Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"llama3-8b-instruct":{"id":"llama3-8b-instruct","name":"Llama 3.1 Instruct (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.198,"output":0.198}},"glm-5.3":{"id":"glm-5.3","name":"GLM5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.95,"output":3.4,"cache_read":0.2}},"anthropic-claude-opus-5":{"id":"anthropic-claude-opus-5","name":"Anthropic Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai-gpt-5.4-nano":{"id":"openai-gpt-5.4-nano","name":"OpenAI GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"anthropic-claude-4.6-sonnet":{"id":"anthropic-claude-4.6-sonnet","name":"Anthropic Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic-claude-haiku-4.5":{"id":"anthropic-claude-haiku-4.5","name":"Anthropic Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":1.5,"cache_read":0.08}},"openai-gpt-image-2":{"id":"openai-gpt-image-2","name":"OpenAI GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":8,"output":30}},"openai-gpt-4o-mini":{"id":"openai-gpt-4o-mini","name":"OpenAI GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"fal-ai/fast-sdxl":{"id":"fal-ai/fast-sdxl","name":"Fast SDXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-07-26","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}},"fal-ai/elevenlabs/tts/multilingual-v2":{"id":"fal-ai/elevenlabs/tts/multilingual-v2","name":"ElevenLabs Multilingual TTS v2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-08-22","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fal-ai/stable-audio-25/text-to-audio":{"id":"fal-ai/stable-audio-25/text-to-audio","name":"Stable Audio 2.5 (Text-to-Audio)","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-08","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fal-ai/flux/schnell":{"id":"fal-ai/flux/schnell","name":"FLUX.1 [schnell]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-01","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}}}},"vivgrid":{"id":"vivgrid","env":["VIVGRID_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.vivgrid.com/v1","name":"Vivgrid","doc":"https://docs.vivgrid.com/models","models":{"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.35,"output":3,"reasoning":3,"cache_read":0.05}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.2,"cache_read":0.3}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.3,"reasoning":0.3,"cache_read":0.03}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.31,"output":1.23,"cache_read":0.01}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.5,"cache_write":12.5}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"viv-fast":{"id":"viv-fast","name":"Viv Fast","description":"Fast coding model","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-09","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":256000},"cost":{"input":0.13,"output":0.4,"cache_read":0.05}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.28,"output":0.42}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1.25,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.15}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"auriko":{"id":"auriko","env":["AURIKO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.auriko.ai/v1","name":"Auriko","doc":"https://docs.auriko.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_write":0.375}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"qwen-3.6-plus":{"id":"qwen-3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_write":0.375}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}}}},"siliconflow-cn":{"id":"siliconflow-cn","env":["SILICONFLOW_CN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.cn/v1","name":"SiliconFlow (China)","doc":"https://cloud.siliconflow.com/models","models":{"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-OCR":{"id":"deepseek-ai/DeepSeek-OCR","name":"deepseek-ai/DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"deepseek-ai/DeepSeek-V4-Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":393000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen/Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.09}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3.5-4B":{"id":"Qwen/Qwen3.5-4B","name":"Qwen/Qwen3.5-4B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.74}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen/Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.32}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen/Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.74}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen/Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen/Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen/Qwen3.6-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"Pro/deepseek-ai/DeepSeek-V3":{"id":"Pro/deepseek-ai/DeepSeek-V3","name":"Pro/deepseek-ai/DeepSeek-V3","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"Pro/deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","name":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"Pro/deepseek-ai/DeepSeek-R1":{"id":"Pro/deepseek-ai/DeepSeek-R1","name":"Pro/deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"Pro/deepseek-ai/DeepSeek-V3.2":{"id":"Pro/deepseek-ai/DeepSeek-V3.2","name":"Pro/deepseek-ai/DeepSeek-V3.2","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"Pro/zai-org/GLM-5.1":{"id":"Pro/zai-org/GLM-5.1","name":"Pro/zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"Pro/zai-org/GLM-5":{"id":"Pro/zai-org/GLM-5","name":"Pro/zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1,"output":3.2}},"Pro/MiniMaxAI/MiniMax-M2.5":{"id":"Pro/MiniMaxAI/MiniMax-M2.5","name":"Pro/MiniMaxAI/MiniMax-M2.5","description":"Frontier MiniMax model for engineering, office tasks, and agentic reasoning","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":131000},"cost":{"input":0.3,"output":1.22}},"Pro/moonshotai/Kimi-K2.5":{"id":"Pro/moonshotai/Kimi-K2.5","name":"Pro/moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"Pro/moonshotai/Kimi-K2.6":{"id":"Pro/moonshotai/Kimi-K2.6","name":"Pro/moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"PaddlePaddle/PaddleOCR-VL-1.5":{"id":"PaddlePaddle/PaddleOCR-VL-1.5","name":"PaddlePaddle/PaddleOCR-VL-1.5","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-29","last_updated":"2026-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0,"output":0}}}},"nova":{"id":"nova","env":["NOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nova.amazon.com/v1","name":"Nova","doc":"https://nova.amazon.com/dev/documentation","models":{"nova-2-lite-v1":{"id":"nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}},"nova-2-pro-v1":{"id":"nova-2-pro-v1","name":"Nova 2 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-01-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}}}},"inceptron":{"id":"inceptron","env":["INCEPTRON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptron.io/v1","name":"Inceptron","doc":"https://docs.inceptron.io","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.13,"output":0.28,"cache_read":0.03,"cache_write":0}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.71,"output":2.35,"cache_read":0.12,"cache_write":0}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.4,"cache_read":0.18,"cache_write":0}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.53,"output":3.39,"cache_read":0.17,"cache_write":0}}}},"vultr":{"id":"vultr","env":["VULTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.vultrinference.com/v1","name":"Vultr","doc":"https://api.vultrinference.com/","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1}},"nvidia/DeepSeek-V3.2-NVFP4":{"id":"nvidia/DeepSeek-V3.2-NVFP4","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":1.65}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","name":"NVIDIA Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.38}},"nvidia/Nemotron-Cascade-2-30B-A3B":{"id":"nvidia/Nemotron-Cascade-2-30B-A3B","name":"NVIDIA Nemotron Cascade 2","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":393216,"output":131072},"cost":{"input":0.85,"output":3.1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":1.2}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.55,"output":1.65}}}},"ollama-cloud":{"id":"ollama-cloud","env":["OLLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ollama.com/v1","name":"Ollama Cloud","doc":"https://docs.ollama.com/cloud","models":{"gpt-oss:20b":{"id":"gpt-oss:20b","name":"gpt-oss:20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.3,"cache_read":0.035}},"deepseek-v4-flash:0731":{"id":"deepseek-v4-flash:0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"minimax-m2.7":{"id":"minimax-m2.7","name":"minimax-m2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kimi-k2.6":{"id":"kimi-k2.6","name":"kimi-k2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":976000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"minimax-m2.5":{"id":"minimax-m2.5","name":"minimax-m2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072}},"minimax-m3":{"id":"minimax-m3","name":"minimax-m3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"qwen3.5:397b":{"id":"qwen3.5:397b","name":"qwen3.5:397b","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"release_date":"2026-02-15","last_updated":"2026-02-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"kimi-k2.7-code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"gpt-oss:120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.014}},"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"nemotron-3-ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.1,"output":3,"cache_read":0.1}},"deepseek-v4-pro:0813":{"id":"deepseek-v4-pro:0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"nemotron-3-nano:30b":{"id":"nemotron-3-nano:30b","name":"nemotron-3-nano:30b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.06,"output":0.24}},"mistral-large-3:675b":{"id":"mistral-large-3:675b","name":"mistral-large-3:675b","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-12-02","last_updated":"2026-01-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"kimi-k3":{"id":"kimi-k3","name":"kimi-k3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"gemma4:31b":{"id":"gemma4:31b","name":"gemma4:31b","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.4,"cache_read":0.05}},"kimi-k2.5":{"id":"kimi-k2.5","name":"kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"glm-5.1":{"id":"glm-5.1","name":"glm-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-03-27","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"deepseek-v4-pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"nemotron-3-super":{"id":"nemotron-3-super","name":"nemotron-3-super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.015,"output":0.6,"cache_read":0.015}}}},"freemodel":{"id":"freemodel","env":["FREEMODEL_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://cc.freemodel.dev/v1","name":"FreeModel","doc":"https://freemodel.dev","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}}}},"iflowcn":{"id":"iflowcn","env":["IFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apis.iflow.cn/v1","name":"iFlow","doc":"https://platform.iflow.cn/en/docs","models":{"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3-235B-A22B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-instruct":{"id":"qwen3-235b-a22b-instruct","name":"Qwen3-235B-A22B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-235b":{"id":"qwen3-235b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi-K2-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL-Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3-Max-Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"kimi-k2":{"id":"kimi-k2","name":"Kimi-K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}}}},"scx-ai":{"id":"scx-ai","env":["SCX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scx.ai/v1","name":"SCX.ai","doc":"https://platform.scx.ai/docs","models":{"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":983616,"output":131072},"cost":{"input":1.815,"output":5.4461,"cache_read":0.17,"cache_write":2.5}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.55,"output":1.784,"cache_read":0.111}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.17,"output":0.55}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}}}},"evroc":{"id":"evroc","env":["EVROC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.think.evroc.com/v1","name":"evroc","doc":"https://docs.evroc.com/products/think/overview.html","models":{"evroc/roc":{"id":"evroc/roc","name":"roc","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":2.875,"output":11.516}},"mistralai/Voxtral-Small-24B-2507":{"id":"mistralai/Voxtral-Small-24B-2507","name":"Voxtral Small 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["audio","text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"mistralai/Mistral-Medium-3.5-128B":{"id":"mistralai/Mistral-Medium-3.5-128B","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.725,"output":6.9}},"nvidia/Llama-3.3-70B-Instruct-FP8":{"id":"nvidia/Llama-3.3-70B-Instruct-FP8","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.15,"output":1.15}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.144,"output":0.575}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1.4375,"output":5.75}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.87,"output":3.5}},"Qwen/Qwen3-Reranker-4B":{"id":"Qwen/Qwen3-Reranker-4B","name":"Qwen3 Reranker 4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.0575,"output":0}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.345,"output":1.38}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":4096},"cost":{"input":0.115,"output":0.115}},"intfloat/multilingual-e5-large-instruct":{"id":"intfloat/multilingual-e5-large-instruct","name":"E5 Multi-Lingual Large Embeddings 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"cost":{"input":0.114,"output":0.114}},"KBLab/kb-whisper-large":{"id":"KBLab/kb-whisper-large","name":"KB Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper 3 Large","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/whisper-large-v3-turbo":{"id":"openai/whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.23,"output":0.92}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.4375,"output":5.75}}}},"echo":{"id":"echo","env":["ECHO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://echo.tracerml.ai/v1","name":"Echo","doc":"https://echo.tracerml.ai/docs/api","models":{"echo":{"id":"echo","name":"Echo","description":"Adaptive model for coding, reasoning, and tool-driven agent workflows through one OpenAI-compatible endpoint","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"beta","cost":{"input":10,"output":50}}}},"aixy":{"id":"aixy","env":["AIXY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aixy-gateway.com/v1","name":"Aixy","doc":"https://docs.aixy-gateway.com/integrations/overview","models":{"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}}}},"impossibl":{"id":"impossibl","env":["IMPOSSIBL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.impossibl.com/v1","name":"Impossibl","doc":"https://impossibl.com/docs/models","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"qwen/qwen3.8-max-preview":{"id":"qwen/qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"tiers":[{"input":1,"output":4,"cache_read":0.2,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1,"output":4,"cache_read":0.2}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.19,"output":0.51,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"fireworks/glm-5.2":{"id":"fireworks/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"fireworks/gpt-oss-20b":{"id":"fireworks/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.3,"cache_read":0.035}},"fireworks/gpt-oss-120b":{"id":"fireworks/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}}}},"llmgateway-providers":{"id":"llmgateway-providers","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"LLM Gateway","doc":"https://llmgateway.io/docs","models":{"atria/atria-dawn-preview":{"id":"atria/atria-dawn-preview","name":"Atria Dawn Preview (Atria)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"vertex-openai/glm-4.7":{"id":"vertex-openai/glm-4.7","name":"GLM-4.7 (Vertex AI (OpenAI-compatible))","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.6,"output":2.2}},"vertex-openai/qwen3-next-80b-a3b-thinking":{"id":"vertex-openai/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking (Vertex AI (OpenAI-compatible))","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/qwen3-next-80b-a3b-instruct":{"id":"vertex-openai/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (Vertex AI (OpenAI-compatible))","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/kimi-k2-thinking":{"id":"vertex-openai/kimi-k2-thinking","name":"Kimi K2 Thinking (Vertex AI (OpenAI-compatible))","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"vertex-openai/deepseek-v3.2":{"id":"vertex-openai/deepseek-v3.2","name":"DeepSeek V3.2 (Vertex AI (OpenAI-compatible))","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"vertex-openai/glm-5":{"id":"vertex-openai/glm-5","name":"GLM-5 (Vertex AI (OpenAI-compatible))","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"vertex-openai/qwen3-235b-a22b-instruct-2507":{"id":"vertex-openai/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Vertex AI (OpenAI-compatible))","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.22,"output":0.88}},"vertex-openai/grok-4-6":{"id":"vertex-openai/grok-4-6","name":"Grok 4.6 (Vertex AI (OpenAI-compatible))","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"vertex-openai/qwen3-coder-480b-a35b-instruct":{"id":"vertex-openai/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (Vertex AI (OpenAI-compatible))","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"vertex-openai/grok-4-20-non-reasoning":{"id":"vertex-openai/grok-4-20-non-reasoning","name":"Grok 4.20 Non-Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"vertex-openai/grok-4-20-reasoning":{"id":"vertex-openai/grok-4-20-reasoning","name":"Grok 4.20 Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"baidu/kimi-k2.6":{"id":"baidu/kimi-k2.6","name":"Kimi K2.6 (Baidu)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"baidu/glm-5.2":{"id":"baidu/glm-5.2","name":"GLM-5.2 (Baidu)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-flash":{"id":"baidu/deepseek-v4-flash","name":"DeepSeek V4 Flash (Baidu)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"baidu/deepseek-v4.1-flash":{"id":"baidu/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Baidu)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"baidu/glm-5":{"id":"baidu/glm-5","name":"GLM-5 (Baidu)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"baidu/glm-5.1":{"id":"baidu/glm-5.1","name":"GLM-5.1 (Baidu)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-pro":{"id":"baidu/deepseek-v4-pro","name":"DeepSeek V4 Pro (Baidu)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.042}},"baidu/glm-5.3":{"id":"baidu/glm-5.3","name":"GLM-5.3 (Baidu)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"aws-mantle/gpt-5.6-sol":{"id":"aws-mantle/gpt-5.6-sol","name":"GPT-5.6 Sol (AWS Mantle)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5}},"aws-mantle/gpt-6-astra":{"id":"aws-mantle/gpt-6-astra","name":"GPT-6 Astra (AWS Mantle)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"aws-mantle/gpt-5.6-luna":{"id":"aws-mantle/gpt-5.6-luna","name":"GPT-5.6 Luna (AWS Mantle)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"aws-mantle/gpt-5.6-terra":{"id":"aws-mantle/gpt-5.6-terra","name":"GPT-5.6 Terra (AWS Mantle)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75}},"gonka24/minimax-m2.7":{"id":"gonka24/minimax-m2.7","name":"MiniMax M2.7 (Gonka24)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.08,"output":0.32,"cache_read":0.017}},"gonka24/deepseek-v4-flash":{"id":"gonka24/deepseek-v4-flash","name":"DeepSeek V4 Flash (Gonka24)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":390000,"output":16384},"cost":{"input":0.051,"output":0.104,"cache_read":0.0097}},"gonka24/glm-5.3-flash":{"id":"gonka24/glm-5.3-flash","name":"GLM-5.3 Flash (Gonka24)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":16384},"cost":{"input":0.07,"output":0.19,"cache_read":0.015}},"embercloud/glm-4.7":{"id":"embercloud/glm-4.7","name":"GLM-4.7 (EmberCloud)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.38,"output":1.98,"cache_read":0.19}},"embercloud/glm-4.5-air":{"id":"embercloud/glm-4.5-air","name":"GLM-4.5 Air (EmberCloud)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"embercloud/glm-5.2":{"id":"embercloud/glm-5.2","name":"GLM-5.2 (EmberCloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"embercloud/qwen3-coder-next":{"id":"embercloud/qwen3-coder-next","name":"Qwen3 Coder Next (EmberCloud)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"embercloud/glm-4.5":{"id":"embercloud/glm-4.5","name":"GLM-4.5 (EmberCloud)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"embercloud/glm-5":{"id":"embercloud/glm-5","name":"GLM-5 (EmberCloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.72,"output":2.3,"cache_read":0.144}},"embercloud/kimi-k2.5":{"id":"embercloud/kimi-k2.5","name":"Kimi K2.5 (EmberCloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"embercloud/glm-5.1":{"id":"embercloud/glm-5.1","name":"GLM-5.1 (EmberCloud)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.931,"output":2.93,"cache_read":0.173}},"embercloud/glm-4.7-flash":{"id":"embercloud/glm-4.7-flash","name":"GLM-4.7 Flash (EmberCloud)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"scx-ai/minimax-m2.7":{"id":"scx-ai/minimax-m2.7","name":"MiniMax M2.7 (SCX.ai (Turbo))","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}},"scx-ai/qwen3-32b":{"id":"scx-ai/qwen3-32b","name":"Qwen3 32B (SCX.ai (Turbo))","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.36,"output":0.87}},"scx-ai/llama-4-maverick-17b-instruct":{"id":"scx-ai/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (SCX.ai (Turbo))","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.53,"output":1.62}},"scx-ai/gemma-4-31b-it":{"id":"scx-ai/gemma-4-31b-it","name":"Gemma 4 31B IT (SCX.ai (Turbo))","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.91}},"scx-ai/gpt-oss-120b":{"id":"scx-ai/gpt-oss-120b","name":"GPT OSS 120B (SCX.ai (Turbo))","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.17,"output":0.55}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.1,"output":0.5}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.15,"output":0.75}},"google-vertex/gemini-3.1-pro-preview":{"id":"google-vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-vertex/gemini-2.5-flash-lite":{"id":"google-vertex/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google Vertex AI)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-vertex/gemini-3.6-flash":{"id":"google-vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3.1-flash-lite":{"id":"google-vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google-vertex/gemini-3.5-flash":{"id":"google-vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-vertex/gemini-3.5-flash-lite":{"id":"google-vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-vertex/gemini-3-flash-preview":{"id":"google-vertex/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-vertex/gemini-3.8-flash":{"id":"google-vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3.7-flash":{"id":"google-vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-2.5-pro":{"id":"google-vertex/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google Vertex AI)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-vertex/gemini-2.5-flash":{"id":"google-vertex/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google Vertex AI)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"quartz/gemini-3.1-pro-preview":{"id":"quartz/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Quartz)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"vertex-anthropic/claude-sonnet-4-6":{"id":"vertex-anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Vertex AI (Anthropic))","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-opus-4-6":{"id":"vertex-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Vertex AI (Anthropic))","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-opus-4-7":{"id":"vertex-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Vertex AI (Anthropic))","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-haiku-4-5":{"id":"vertex-anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Vertex AI (Anthropic))","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"vertex-anthropic/claude-sonnet-4-5":{"id":"vertex-anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Vertex AI (Anthropic))","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-sonnet-5":{"id":"vertex-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Vertex AI (Anthropic))","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"vertex-anthropic/claude-opus-4-5-20251101":{"id":"vertex-anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Vertex AI (Anthropic))","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5 (Xiaomi)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro (Xiaomi)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1 (MiniMax)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.27,"output":1.1}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2 (MiniMax)","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed (MiniMax)","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7 (MiniMax)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5 (MiniMax)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3 (MiniMax)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"minimax/minimax-text-01":{"id":"minimax/minimax-text-01","name":"MiniMax Text 01 (MiniMax)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max (Alibaba Cloud)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus (Alibaba Cloud)","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":66000},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"alibaba/qwen35-397b-a17b":{"id":"alibaba/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3-coder-flash":{"id":"alibaba/qwen3-coder-flash","name":"Qwen3 Coder Flash (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"alibaba/qwen-max":{"id":"alibaba/qwen-max","name":"Qwen Max (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen3.6 Plus (Alibaba Cloud)","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"alibaba/qwen-flash":{"id":"alibaba/qwen-flash","name":"Qwen Flash (Alibaba Cloud)","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"alibaba/glm-5.2":{"id":"alibaba/glm-5.2","name":"GLM-5.2 (Alibaba Cloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28}},"alibaba/deepseek-v4-flash":{"id":"alibaba/deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"alibaba/qwen3-vl-plus":{"id":"alibaba/qwen3-vl-plus","name":"Qwen3 VL Plus (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"cache_read":0.04,"cache_write":0.25}},"alibaba/deepseek-v4.1-flash":{"id":"alibaba/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Alibaba Cloud)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"alibaba/qwen-coder-plus":{"id":"alibaba/qwen-coder-plus","name":"Qwen Coder Plus (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen3.7 Flash (Alibaba Cloud)","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"alibaba/kimi-k3":{"id":"alibaba/kimi-k3","name":"Kimi K3 (Alibaba Cloud)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (Alibaba Cloud)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.375,"output":2.25}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max (Alibaba Cloud)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32800},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"alibaba/qwen3-vl-flash":{"id":"alibaba/qwen3-vl-flash","name":"Qwen3 VL Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"alibaba/qwen-plus":{"id":"alibaba/qwen-plus","name":"Qwen Plus (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen3.6-flash":{"id":"alibaba/qwen3.6-flash","name":"Qwen3.6 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen3.8 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/qwen3.6-max-preview":{"id":"alibaba/qwen3.6-max-preview","name":"Qwen3.6 Max Preview (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13}},"alibaba/glm-5":{"id":"alibaba/glm-5","name":"GLM-5 (Alibaba Cloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max (Alibaba Cloud)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/kimi-k2.5":{"id":"alibaba/kimi-k2.5","name":"Kimi K2.5 (Alibaba Cloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.574,"output":3.011}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus (Alibaba Cloud)","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen-omni-turbo":{"id":"alibaba/qwen-omni-turbo","name":"Qwen Omni Turbo (Alibaba Cloud)","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.2,"output":0.8}},"alibaba/deepseek-v4-pro":{"id":"alibaba/deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"alibaba/glm-5.3":{"id":"alibaba/glm-5.3","name":"GLM-5.3 (Alibaba Cloud)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28}},"alibaba/qwen-plus-latest":{"id":"alibaba/qwen-plus-latest","name":"Qwen Plus Latest (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-09","last_updated":"2024-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"runpod/kimi-k3":{"id":"runpod/kimi-k3","name":"Kimi K3 (Runpod)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"scx-ai-gp/glm-5.2":{"id":"scx-ai-gp/glm-5.2","name":"GLM-5.2 (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.88,"output":2.55,"cache_read":0.16}},"scx-ai-gp/kimi-k2.7-code":{"id":"scx-ai-gp/kimi-k2.7-code","name":"Kimi K2.7 Code (SCX.ai)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"scx-ai-gp/deepseek-v4.1-flash":{"id":"scx-ai-gp/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (SCX.ai)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.01}},"scx-ai-gp/kimi-k3":{"id":"scx-ai-gp/kimi-k3","name":"Kimi K3 (SCX.ai)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3.5,"output":18,"cache_read":0.35}},"scx-ai-gp/glm-5.2-fast":{"id":"scx-ai-gp/glm-5.2-fast","name":"GLM-5.2 Turbo (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"scx-ai-gp/glm-5.3-flash":{"id":"scx-ai-gp/glm-5.3-flash","name":"GLM-5.3 Flash (SCX.ai)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.088,"output":0.25,"cache_read":0.025}},"scx-ai-gp/qwen3.8-max":{"id":"scx-ai-gp/qwen3.8-max","name":"Qwen3.8 Max (SCX.ai)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"scx-ai-gp/glm-5.3":{"id":"scx-ai-gp/glm-5.3","name":"GLM-5.3 (SCX.ai)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"aws-bedrock/claude-sonnet-4-6":{"id":"aws-bedrock/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (AWS Bedrock)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/llama-4-scout-17b-instruct":{"id":"aws-bedrock/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (AWS Bedrock)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.17,"output":0.66}},"aws-bedrock/claude-opus-5":{"id":"aws-bedrock/claude-opus-5","name":"Claude Opus 5 (AWS Bedrock)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-opus-4-1-20250805":{"id":"aws-bedrock/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"aws-bedrock/claude-fable-5-1":{"id":"aws-bedrock/claude-fable-5-1","name":"Claude Fable 5.1 (AWS Bedrock)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"aws-bedrock/claude-opus-4-6":{"id":"aws-bedrock/claude-opus-4-6","name":"Claude Opus 4.6 (AWS Bedrock)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-sonnet-4-5-20250929":{"id":"aws-bedrock/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/claude-opus-4-7":{"id":"aws-bedrock/claude-opus-4-7","name":"Claude Opus 4.7 (AWS Bedrock)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-haiku-4-5-20251001":{"id":"aws-bedrock/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (AWS Bedrock)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/claude-fable-5":{"id":"aws-bedrock/claude-fable-5","name":"Claude Fable 5 (AWS Bedrock)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"aws-bedrock/llama-4-maverick-17b-instruct":{"id":"aws-bedrock/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (AWS Bedrock)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.24,"output":0.97}},"aws-bedrock/grok-4-3":{"id":"aws-bedrock/grok-4-3","name":"Grok 4.3 (AWS Bedrock)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"aws-bedrock/claude-haiku-4-5":{"id":"aws-bedrock/claude-haiku-4-5","name":"Claude Haiku 4.5 (AWS Bedrock)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/claude-sonnet-4-5":{"id":"aws-bedrock/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/llama-3.1-70b-instruct":{"id":"aws-bedrock/llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct (AWS Bedrock)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.72,"output":0.72}},"aws-bedrock/grok-4-6":{"id":"aws-bedrock/grok-4-6","name":"Grok 4.6 (AWS Bedrock)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"aws-bedrock/claude-opus-4-8":{"id":"aws-bedrock/claude-opus-4-8","name":"Claude Opus 4.8 (AWS Bedrock)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-sonnet-5":{"id":"aws-bedrock/claude-sonnet-5","name":"Claude Sonnet 5 (AWS Bedrock)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"aws-bedrock/claude-opus-4-5-20251101":{"id":"aws-bedrock/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Anthropic)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5 (Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1 (Anthropic)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (Anthropic)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5 (Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Anthropic)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Anthropic)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"canopywave/kimi-k2.6":{"id":"canopywave/kimi-k2.6","name":"Kimi K2.6 (CanopyWave)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"canopywave/glm-5.2":{"id":"canopywave/glm-5.2","name":"GLM-5.2 (CanopyWave)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"canopywave/deepseek-v4-flash":{"id":"canopywave/deepseek-v4-flash","name":"DeepSeek V4 Flash (CanopyWave)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"canopywave/kimi-k3":{"id":"canopywave/kimi-k3","name":"Kimi K3 (CanopyWave)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"canopywave/deepseek-v4-pro":{"id":"canopywave/deepseek-v4-pro","name":"DeepSeek V4 Pro (CanopyWave)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.74,"output":3.48,"cache_read":0.01}},"together-ai/glm-4.7":{"id":"together-ai/glm-4.7","name":"GLM-4.7 (Together AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.45,"output":2}},"together-ai/minimax-m3":{"id":"together-ai/minimax-m3","name":"MiniMax M3 (Together AI)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"together-ai/deepseek-v4-flash":{"id":"together-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash (Together AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"together-ai/deepseek-v4.1-flash":{"id":"together-ai/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Together AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"together-ai/kimi-k3":{"id":"together-ai/kimi-k3","name":"Kimi K3 (Together AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":3,"output":15,"cache_read":0.3}},"together-ai/deepseek-v4-pro":{"id":"together-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro (Together AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":163840},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together-ai/gpt-oss-120b":{"id":"together-ai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3 (Meta)","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2 (Meta)","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1 (Meta)","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"google-ai-studio/gemini-pro-latest":{"id":"google-ai-studio/gemini-pro-latest","name":"Gemini Pro Latest (Google AI Studio)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google-ai-studio/gemini-3.1-pro-preview":{"id":"google-ai-studio/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google AI Studio)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-ai-studio/gemini-2.5-flash-lite":{"id":"google-ai-studio/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google AI Studio)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-ai-studio/gemini-3.6-flash":{"id":"google-ai-studio/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3.1-flash-lite":{"id":"google-ai-studio/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google AI Studio)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google-ai-studio/gemini-3.5-flash":{"id":"google-ai-studio/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-ai-studio/gemini-3.5-flash-lite":{"id":"google-ai-studio/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-ai-studio/gemini-3-flash-preview":{"id":"google-ai-studio/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google AI Studio)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-ai-studio/gemini-3.8-flash":{"id":"google-ai-studio/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google AI Studio)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3.7-flash":{"id":"google-ai-studio/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google AI Studio)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-2.5-pro":{"id":"google-ai-studio/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google AI Studio)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-ai-studio/gemini-2.5-flash":{"id":"google-ai-studio/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google AI Studio)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"bytedance/glm-4.7":{"id":"bytedance/glm-4.7","name":"GLM-4.7 (ByteDance)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"bytedance/seed-1-8-251228":{"id":"bytedance/seed-1-8-251228","name":"Seed 1.8 (251228) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/glm-5.2":{"id":"bytedance/glm-5.2","name":"GLM-5.2 (ByteDance)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"bytedance/deepseek-v4-flash":{"id":"bytedance/deepseek-v4-flash","name":"DeepSeek V4 Flash (ByteDance)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"bytedance/seed-1-6-flash-250715":{"id":"bytedance/seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"bytedance/deepseek-v3.2":{"id":"bytedance/deepseek-v3.2","name":"DeepSeek V3.2 (ByteDance)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.28,"output":0.42,"cache_read":0.056}},"bytedance/seed-1-6-250615":{"id":"bytedance/seed-1-6-250615","name":"Seed 1.6 (250615) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/deepseek-v4-pro":{"id":"bytedance/deepseek-v4-pro","name":"DeepSeek V4 Pro (ByteDance)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"bytedance/gpt-oss-120b":{"id":"bytedance/gpt-oss-120b","name":"GPT OSS 120B (ByteDance)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.1,"output":0.5,"cache_read":0.02}},"bytedance/seed-1-6-250915":{"id":"bytedance/seed-1-6-250915","name":"Seed 1.6 (250915) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"GLM-4.7 (NovitaAI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"novita/qwen3.7-max":{"id":"novita/qwen3.7-max","name":"Qwen3.7 Max (NovitaAI)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"novita/gemma-4-26b-a4b-it":{"id":"novita/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (NovitaAI)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"novita/llama-4-scout-17b-instruct":{"id":"novita/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (NovitaAI)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"novita/qwen35-397b-a17b":{"id":"novita/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"novita/qwen3-235b-a22b-thinking-2507":{"id":"novita/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507 (NovitaAI)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6 (NovitaAI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"MiniMax M2.1 (NovitaAI)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"GLM-4.6V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"novita/qwen3-next-80b-a3b-instruct":{"id":"novita/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (NovitaAI)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"novita/ling-3.0-flash":{"id":"novita/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (NovitaAI)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"novita/qwen3.8-27b":{"id":"novita/qwen3.8-27b","name":"Qwen3.8 27B (NovitaAI)","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"novita/qwen3-235b-a22b-fp8":{"id":"novita/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8 (NovitaAI)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"novita/minimax-m2.7":{"id":"novita/minimax-m2.7","name":"MiniMax M2.7 (NovitaAI)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi K2.6 (NovitaAI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"novita/glm-5.2":{"id":"novita/glm-5.2","name":"GLM-5.2 (NovitaAI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/minimax-m2.5":{"id":"novita/minimax-m2.5","name":"MiniMax M2.5 (NovitaAI)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/deepseek-v4-flash":{"id":"novita/deepseek-v4-flash","name":"DeepSeek V4 Flash (NovitaAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"novita/kimi-k2.7-code":{"id":"novita/kimi-k2.7-code","name":"Kimi K2.7 Code (NovitaAI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"novita/llama-3.2-3b-instruct":{"id":"novita/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"novita/deepseek-v4.1-flash":{"id":"novita/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (NovitaAI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"novita/hy3":{"id":"novita/hy3","name":"Hy3 (NovitaAI)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"novita/qwen3-coder-30b-a3b-instruct":{"id":"novita/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct (NovitaAI)","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"novita/ernie-4.5-vl-424b-a47b":{"id":"novita/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"novita/kimi-k3":{"id":"novita/kimi-k3","name":"Kimi K3 (NovitaAI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek V3.2 (NovitaAI)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"novita/qwen3.6-35b-a3b":{"id":"novita/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (NovitaAI)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.248,"output":1.485}},"novita/qwen3-max":{"id":"novita/qwen3-max","name":"Qwen3 Max (NovitaAI)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38}},"novita/glm-5.3-flash":{"id":"novita/glm-5.3-flash","name":"GLM-5.3 Flash (NovitaAI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"novita/qwen3-vl-30b-a3b-instruct":{"id":"novita/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (NovitaAI)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"novita/llama-4-maverick-17b-instruct":{"id":"novita/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (NovitaAI)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"novita/glm-4.5v":{"id":"novita/glm-4.5v","name":"GLM-4.5V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"novita/qwen3.8-flash":{"id":"novita/qwen3.8-flash","name":"Qwen3.8 Flash (NovitaAI)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"novita/kimi-k2":{"id":"novita/kimi-k2","name":"Kimi K2 (NovitaAI)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"novita/gemma-4-31b-it":{"id":"novita/gemma-4-31b-it","name":"Gemma 4 31B IT (NovitaAI)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5 (NovitaAI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/qwen3.8-max":{"id":"novita/qwen3.8-max","name":"Qwen3.8 Max (NovitaAI)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"novita/glm-5.1":{"id":"novita/glm-5.1","name":"GLM-5.1 (NovitaAI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"novita/qwen3-vl-235b-a22b-thinking":{"id":"novita/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking (NovitaAI)","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"novita/qwen3-vl-235b-a22b-instruct":{"id":"novita/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (NovitaAI)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"novita/qwen3-235b-a22b-instruct-2507":{"id":"novita/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (NovitaAI)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"novita/qwen3-coder-480b-a35b-instruct":{"id":"novita/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (NovitaAI)","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"novita/glm-5.3":{"id":"novita/glm-5.3","name":"GLM-5.3 (NovitaAI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/llama-3.3-70b-instruct":{"id":"novita/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (NovitaAI)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"novita/llama-3-70b-instruct":{"id":"novita/llama-3-70b-instruct","name":"Llama 3 70B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"novita/mimo-v2.5":{"id":"novita/mimo-v2.5","name":"MiMo V2.5 (NovitaAI)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.168,"output":0.336,"cache_read":0.0034,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"novita/mimo-v2.5-pro":{"id":"novita/mimo-v2.5-pro","name":"MiMo V2.5 Pro (NovitaAI)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"ranoai/deepseek-v4-flash":{"id":"ranoai/deepseek-v4-flash","name":"DeepSeek V4 Flash (RanoAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"inference.net/llama-3.2-11b-instruct":{"id":"inference.net/llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct (Inference.net)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.33}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra (Sakana AI)","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max (Sakana AI)","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2.0":{"id":"sakana/fugu-ultra-v2.0","name":"Fugu Ultra v2.0 (Sakana AI)","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"deepinfra/gemma-4-26b-a4b-it":{"id":"deepinfra/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (DeepInfra)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"deepinfra/qwen3.5-9b":{"id":"deepinfra/qwen3.5-9b","name":"Qwen3.5 9B (DeepInfra)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.15}},"deepinfra/ling-3.0-flash":{"id":"deepinfra/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (DeepInfra)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"deepinfra/deepseek-v4-flash":{"id":"deepinfra/deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepInfra)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.08,"output":0.18,"cache_read":0.016}},"deepinfra/deepseek-v4.1-flash":{"id":"deepinfra/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (DeepInfra)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepinfra/hy3":{"id":"deepinfra/hy3","name":"Hy3 (DeepInfra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"deepinfra/deepseek-v3.2":{"id":"deepinfra/deepseek-v3.2","name":"DeepSeek V3.2 (DeepInfra)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":65536},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepinfra/nemotron-3-ultra-550b":{"id":"deepinfra/nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B (DeepInfra)","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/qwen3-vl-30b-a3b-instruct":{"id":"deepinfra/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (DeepInfra)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":0.6}},"deepinfra/gemma-4-31b-it":{"id":"deepinfra/gemma-4-31b-it","name":"Gemma 4 31B IT (DeepInfra)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"deepinfra/glm-5.1":{"id":"deepinfra/glm-5.1","name":"GLM-5.1 (DeepInfra)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":65536},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"deepinfra/qwen3-vl-235b-a22b-instruct":{"id":"deepinfra/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (DeepInfra)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"deepinfra/deepseek-v4-pro":{"id":"deepinfra/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepInfra)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/mimo-v2.5":{"id":"deepinfra/mimo-v2.5","name":"MiMo V2.5 (DeepInfra)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"deepinfra/mimo-v2.5-pro":{"id":"deepinfra/mimo-v2.5-pro","name":"MiMo V2.5 Pro (DeepInfra)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"azure-ai-foundry/grok-4-1-fast-non-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-1-fast-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-3":{"id":"azure-ai-foundry/grok-4-3","name":"Grok 4.3 (Azure AI Foundry)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":8192},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed (Moonshot AI)","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6 (Moonshot AI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code (Moonshot AI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3 (Moonshot AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5 (Moonshot AI)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"consensusprotocol/Qwen3.8-27B":{"id":"consensusprotocol/Qwen3.8-27B","name":"Qwen3.8 27B (Consensus Protocol)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.08,"output":0.35,"cache_read":0.05}},"consensusprotocol/deepseek-v4-flash":{"id":"consensusprotocol/deepseek-v4-flash","name":"DeepSeek V4 Flash (Consensus Protocol)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"consensusprotocol/gpt-oss-20b":{"id":"consensusprotocol/gpt-oss-20b","name":"GPT OSS 20B (Consensus Protocol)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"consensusprotocol/deepseek-v4.1-flash":{"id":"consensusprotocol/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Consensus Protocol)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.005}},"consensusprotocol/glm-5.3-flash":{"id":"consensusprotocol/glm-5.3-flash","name":"GLM-5.3 Flash (Consensus Protocol)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.25,"cache_read":0.02}},"consensusprotocol/gemma-4-31b-it":{"id":"consensusprotocol/gemma-4-31b-it","name":"Gemma 4 31B IT (Consensus Protocol)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"azure/gpt-5-nano":{"id":"azure/gpt-5-nano","name":"GPT-5 Nano (Azure)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"azure/gpt-4.1-nano":{"id":"azure/gpt-4.1-nano","name":"GPT-4.1 Nano (Azure)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":1.25,"output":10}},"azure/gpt-5.6-sol":{"id":"azure/gpt-5.6-sol","name":"GPT-5.6 Sol (Azure)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-6-astra":{"id":"azure/gpt-6-astra","name":"GPT-6 Astra (Azure)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure/gpt-5.2-pro":{"id":"azure/gpt-5.2-pro","name":"GPT-5.2 Pro (Azure)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"azure/gpt-4.1-mini":{"id":"azure/gpt-4.1-mini","name":"GPT-4.1 Mini (Azure)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"azure/gpt-5.4":{"id":"azure/gpt-5.4","name":"GPT-5.4 (Azure)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"azure/gpt-4-turbo":{"id":"azure/gpt-4-turbo","name":"GPT-4 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"azure/gpt-5.1":{"id":"azure/gpt-5.1","name":"GPT-5.1 (Azure)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/o1":{"id":"azure/o1","name":"o1 (Azure)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"azure/gpt-4o":{"id":"azure/gpt-4o","name":"GPT-4o (Azure)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"azure/gpt-5.6-luna":{"id":"azure/gpt-5.6-luna","name":"GPT-5.6 Luna (Azure)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"azure/gpt-5.3-codex":{"id":"azure/gpt-5.3-codex","name":"GPT-5.3 Codex (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-4.1":{"id":"azure/gpt-4.1","name":"GPT-4.1 (Azure)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-5.4-nano":{"id":"azure/gpt-5.4-nano","name":"GPT-5.4 Nano (Azure)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"azure/gpt-5.4-mini":{"id":"azure/gpt-5.4-mini","name":"GPT-5.4 Mini (Azure)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"azure/gpt-3.5-turbo":{"id":"azure/gpt-3.5-turbo","name":"GPT-3.5 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"azure/gpt-5-mini":{"id":"azure/gpt-5-mini","name":"GPT-5 Mini (Azure)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-oss-120b":{"id":"azure/gpt-oss-120b","name":"GPT OSS 120B (Azure)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"azure/gpt-5.4-pro":{"id":"azure/gpt-5.4-pro","name":"GPT-5.4 Pro (Azure)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"azure/gpt-5.6-terra":{"id":"azure/gpt-5.6-terra","name":"GPT-5.6 Terra (Azure)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"azure/gpt-4":{"id":"azure/gpt-4","name":"GPT-4 (Azure)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"azure/gpt-5.2":{"id":"azure/gpt-5.2","name":"GPT-5.2 (Azure)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5":{"id":"azure/gpt-5","name":"GPT-5 (Azure)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/o4-mini":{"id":"azure/o4-mini","name":"o4 Mini (Azure)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"azure/o3-mini":{"id":"azure/o3-mini","name":"o3 Mini (Azure)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"azure/o3":{"id":"azure/o3","name":"o3 (Azure)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-5.5":{"id":"azure/gpt-5.5","name":"GPT-5.5 (Azure)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (DeepSeek)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano (OpenAI)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano (OpenAI)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro (OpenAI)","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":1.25,"output":5}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol (OpenAI)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro (OpenAI)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini (OpenAI)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4 (OpenAI)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":2.5,"output":10}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1 (OpenAI)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1 (OpenAI)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o (OpenAI)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna (OpenAI)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex (OpenAI)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini (OpenAI)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1 (OpenAI)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano (OpenAI)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro (OpenAI)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini (OpenAI)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini (OpenAI)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro (OpenAI)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra (OpenAI)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4 (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2 (OpenAI)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5 (OpenAI)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini (OpenAI)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini (OpenAI)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3 (OpenAI)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5 (OpenAI)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"meta-contributor/muse-spark-1.2-contributor":{"id":"meta-contributor/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta-contributor/muse-spark-1.3-contributor":{"id":"meta-contributor/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"tencent/kimi-k2.7-code-highspeed":{"id":"tencent/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed (Tencent Cloud)","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"tencent/minimax-m2.7":{"id":"tencent/minimax-m2.7","name":"MiniMax M2.7 (Tencent Cloud)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"tencent/kimi-k2.6":{"id":"tencent/kimi-k2.6","name":"Kimi K2.6 (Tencent Cloud)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.858,"output":3.566,"cache_read":0.145}},"tencent/glm-5.2":{"id":"tencent/glm-5.2","name":"GLM-5.2 (Tencent Cloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"tencent/minimax-m3":{"id":"tencent/minimax-m3","name":"MiniMax M3 (Tencent Cloud)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"tencent/deepseek-v4-flash":{"id":"tencent/deepseek-v4-flash","name":"DeepSeek V4 Flash (Tencent Cloud)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"tencent/kimi-k2.7-code":{"id":"tencent/kimi-k2.7-code","name":"Kimi K2.7 Code (Tencent Cloud)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3 (Tencent Cloud)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 Preview (Tencent Cloud)","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-plus":{"id":"tencent/hy-mt2-plus","name":"Hy-MT2 Plus (Tencent Cloud)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/kimi-k3":{"id":"tencent/kimi-k3","name":"Kimi K3 (Tencent Cloud)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"tencent/glm-5":{"id":"tencent/glm-5","name":"GLM-5 (Tencent Cloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"tencent/glm-5.1":{"id":"tencent/glm-5.1","name":"GLM-5.1 (Tencent Cloud)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"tencent/deepseek-v4-pro":{"id":"tencent/deepseek-v4-pro","name":"DeepSeek V4 Pro (Tencent Cloud)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.00363}},"tencent/glm-5-turbo":{"id":"tencent/glm-5-turbo","name":"GLM-5 Turbo (Tencent Cloud)","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"tencent/glm-5v-turbo":{"id":"tencent/glm-5v-turbo","name":"GLM-5V Turbo (Tencent Cloud)","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"tencent/mimo-v2.5-pro":{"id":"tencent/mimo-v2.5-pro","name":"MiMo V2.5 Pro (Tencent Cloud)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok 4 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-7":{"id":"xai/grok-4-7","name":"Grok 4.7 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4-5":{"id":"xai/grok-4-5","name":"Grok 4.5 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-build-0-1":{"id":"xai/grok-build-0-1","name":"Grok Build 0.1 (xAI)","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4-3":{"id":"xai/grok-4-3","name":"Grok 4.3 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-20-beta-0309-reasoning":{"id":"xai/grok-4-20-beta-0309-reasoning","name":"Grok 4.20 Beta Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-6":{"id":"xai/grok-4-6","name":"Grok 4.6 (xAI)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4-20-beta-0309-non-reasoning":{"id":"xai/grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 Beta Non-Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7 (Z AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air (Z AI)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6 (Z AI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-4.6v-flashx":{"id":"zai/glm-4.6v-flashx","name":"GLM-4.6V FlashX (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2 (Z AI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.5-x":{"id":"zai/glm-4.5-x","name":"GLM-4.5 X (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"zai/glm-4.5-airx":{"id":"zai/glm-4.5-airx","name":"GLM-4.5 AirX (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash (Z AI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5 (Z AI)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM-4.5V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX (Z AI)","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5 (Z AI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-4-32b-0414-128k":{"id":"zai/glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k) (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1 (Z AI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3 (Z AI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"azure-anthropic/claude-opus-5":{"id":"azure-anthropic/claude-opus-5","name":"Claude Opus 5 (Azure Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-opus-4-6":{"id":"azure-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Azure Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-opus-4-7":{"id":"azure-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Azure Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-fable-5":{"id":"azure-anthropic/claude-fable-5","name":"Claude Fable 5 (Azure Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure-anthropic/claude-opus-4-8":{"id":"azure-anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Azure Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-sonnet-5":{"id":"azure-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Azure Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"fireworks/deepseek-v4-flash":{"id":"fireworks/deepseek-v4-flash","name":"DeepSeek V4 Flash (Fireworks AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks/deepseek-v4.1-flash":{"id":"fireworks/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Fireworks AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks/kimi-k3":{"id":"fireworks/kimi-k3","name":"Kimi K3 (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":3,"output":15,"cache_read":0.3}},"fireworks/kimi-k3-fast":{"id":"fireworks/kimi-k3-fast","name":"Kimi K3 Fast (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"fireworks/deepseek-v4-pro":{"id":"fireworks/deepseek-v4-pro","name":"DeepSeek V4 Pro (Fireworks AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"mistral/ministral-14b-2512":{"id":"mistral/ministral-14b-2512","name":"Ministral 14B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.2}},"mistral/codestral-2508":{"id":"mistral/codestral-2508","name":"Codestral (Mistral AI)","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"mistral/mistral-small-2506":{"id":"mistral/mistral-small-2506","name":"Mistral Small 3.2 (Mistral AI)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2 (Mistral AI)","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3 (Mistral AI)","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/ministral-3b-2512":{"id":"mistral/ministral-3b-2512","name":"Ministral 3B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large Latest (Mistral AI)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"mistral/ministral-8b-2512":{"id":"mistral/ministral-8b-2512","name":"Ministral 8B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.15,"output":0.15}},"cerebras/glm-4.7":{"id":"cerebras/glm-4.7","name":"GLM-4.7 (Cerebras)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":2.25,"output":2.75}},"cerebras/gemma-4-31b-it":{"id":"cerebras/gemma-4-31b-it","name":"Gemma 4 31B IT (Cerebras)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.99,"output":1.49}},"cerebras/qwen3-235b-a22b-instruct-2507":{"id":"cerebras/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Cerebras)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.6,"output":1.2}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}},"cerebras/llama-3.3-70b-instruct":{"id":"cerebras/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (Cerebras)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.85,"output":1.2}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar (Perplexity)","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro (Perplexity)","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro (Perplexity)","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"runware/kimi-k2.6":{"id":"runware/kimi-k2.6","name":"Kimi K2.6 (Runware)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"runware/glm-5.2":{"id":"runware/glm-5.2","name":"GLM-5.2 (Runware)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"runware/deepseek-v4-flash":{"id":"runware/deepseek-v4-flash","name":"DeepSeek V4 Flash (Runware)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"runware/deepseek-v4.1-flash":{"id":"runware/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Runware)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.01}},"runware/kimi-k3":{"id":"runware/kimi-k3","name":"Kimi K3 (Runware)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"runware/glm-5.3-flash":{"id":"runware/glm-5.3-flash","name":"GLM-5.3 Flash (Runware)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"runware/gemma-4-31b-it":{"id":"runware/gemma-4-31b-it","name":"Gemma 4 31B IT (Runware)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.102,"output":0.297,"cache_read":0.012}},"runware/deepseek-v4-pro":{"id":"runware/deepseek-v4-pro","name":"DeepSeek V4 Pro (Runware)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.961,"output":1.922,"cache_read":0.079}},"runware/gpt-oss-120b":{"id":"runware/gpt-oss-120b","name":"GPT OSS 120B (Runware)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"runware/glm-5.3":{"id":"runware/glm-5.3","name":"GLM-5.3 (Runware)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}}}},"llama":{"id":"llama","env":["LLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llama.com/compat/v1/","name":"Llama","doc":"https://llama.developer.meta.com/docs/models","models":{"cerebras-llama-4-scout-17b-16e-instruct":{"id":"cerebras-llama-4-scout-17b-16e-instruct","name":"Cerebras-Llama-4-Scout-17B-16E-Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"groq-llama-4-maverick-17b-128e-instruct":{"id":"groq-llama-4-maverick-17b-128e-instruct","name":"Groq-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-4-scout-17b-16e-instruct-fp8":{"id":"llama-4-scout-17b-16e-instruct-fp8","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"cerebras-llama-4-maverick-17b-128e-instruct":{"id":"cerebras-llama-4-maverick-17b-128e-instruct","name":"Cerebras-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-8b-instruct":{"id":"llama-3.3-8b-instruct","name":"Llama-3.3-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}}}},"alibaba-token-plan":{"id":"alibaba-token-plan","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/token-plan-overview","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}}}},"neuralwatt":{"id":"neuralwatt","env":["NEURALWATT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.neuralwatt.com/v1","name":"Neuralwatt","doc":"https://portal.neuralwatt.com/docs","models":{"glm-5.2-short-fast-flex":{"id":"glm-5.2-short-fast-flex","name":"GLM 5.2 Short Fast Flex","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.3-flash-flex":{"id":"glm-5.3-flash-flex","name":"GLM-5.3 Flash Flex","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.0975,"output":0.325,"cache_read":0.0195}},"glm-5.3-flex":{"id":"glm-5.3-flex","name":"GLM 5.3 Flex","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"qwen-3.8-27b-flex":{"id":"qwen-3.8-27b-flex","name":"Qwen3.8 27B Flex","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":131072},"cost":{"input":0.2925,"output":2.08,"cache_read":0.1625}},"glm-5.2-short-flex":{"id":"glm-5.2-short-flex","name":"GLM 5.2 Short Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.2-flex":{"id":"glm-5.2-flex","name":"GLM 5.2 Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"deepseek-v4-flash-speed":{"id":"deepseek-v4-flash-speed","name":"DeepSeek V4 Flash (Speed)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.7-code-flex":{"id":"kimi-k2.7-code-flex","name":"Kimi K2.7 Code Flex","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.6175,"output":2.6,"cache_read":0.06175}},"kimi-k2.7-code-fast":{"id":"kimi-k2.7-code-fast","name":"Kimi K2.7 Code Fast","description":"Kimi K2.7 Code with reasoning capped to a short budget for lower latency; reasoning cannot be disabled on this model","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"glm-5.2-short-fast":{"id":"glm-5.2-short-fast","name":"GLM 5.2 Short Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi K3 with thinking disabled for low-latency tool calling, vision, and JSON work","family":"kimi-k3","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3-flex":{"id":"kimi-k3-flex","name":"Kimi K3 Flex","description":"Kimi K3 on the flex tier: discounted, best-effort latency, requests may be held under load","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.95,"output":9.75,"cache_read":0.195}},"qwen3.6-35b-fast":{"id":"qwen3.6-35b-fast","name":"Qwen3.6 35B Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"gemma-4-31b":{"id":"gemma-4-31b","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":16384},"cost":{"input":0.144,"output":0.42,"cache_read":0.0144}},"qwen3.6-35b-flex":{"id":"qwen3.6-35b-flex","name":"Qwen3.6 35B Flex","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.1885,"output":0.7475,"cache_read":0.01885}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"status":"deprecated","cost":{"input":1,"output":3,"cache_read":0.1}},"deepseek-v4-flash-flex":{"id":"deepseek-v4-flash-flex","name":"DeepSeek V4 Flash Flex","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.091,"output":0.182,"cache_read":0.0182}},"deepseek-v4.1-flash-flex":{"id":"deepseek-v4.1-flash-flex","name":"DeepSeek V4.1 Flash Flex","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.0975,"output":0.39,"cache_read":0.00975}},"glm-5.3":{"id":"glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.25}},"glm-5.2-short":{"id":"glm-5.2-short","name":"GLM 5.2 Short","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}}}},"abliteration-ai":{"id":"abliteration-ai","env":["ABLIT_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.abliteration.ai/v1","name":"abliteration.ai","doc":"https://docs.abliteration.ai/models","models":{"abliterated-model-large":{"id":"abliterated-model-large","name":"Abliterated Model Large","description":"GLM-5.2 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-25","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliterated-model-large-v2":{"id":"abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"GLM-5.3 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliterated-model":{"id":"abliterated-model","name":"Abliterated Model","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-06","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":150000,"input":150000,"output":8192},"cost":{"input":3,"output":3,"cache_read":0.3}}}},"clarifai":{"id":"clarifai","env":["CLARIFAI_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://api.clarifai.com/v2/ext/openai/v1","name":"Clarifai","doc":"https://docs.clarifai.com/compute/inference/","models":{"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct":{"id":"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.11458,"output":0.74812}},"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.5}},"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.36,"output":1.3}},"clarifai/main/models/mm-poly-8b":{"id":"clarifai/main/models/mm-poly-8b","name":"MM Poly 8B","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"mm-poly","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.658,"output":1.11}},"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR":{"id":"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR","name":"DeepSeek OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"deepseek","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.2,"output":0.7}},"mistralai/completion/models/Ministral-3-14B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-14B-Reasoning-2512","name":"Ministral 3 14B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-01","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":1.7}},"mistralai/completion/models/Ministral-3-3B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-3B-Reasoning-2512","name":"Ministral 3 3B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.039,"output":0.54825}},"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput":{"id":"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput","name":"MiniMax-M2.5 High Throughput","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"openai/chat-completion/models/gpt-oss-120b-high-throughput":{"id":"openai/chat-completion/models/gpt-oss-120b-high-throughput","name":"GPT OSS 120B High Throughput","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.36}},"openai/chat-completion/models/gpt-oss-20b":{"id":"openai/chat-completion/models/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.045,"output":0.18}},"moonshotai/chat-completion/models/Kimi-K2_6":{"id":"moonshotai/chat-completion/models/Kimi-K2_6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"arcee_ai/AFM/models/trinity-mini":{"id":"arcee_ai/AFM/models/trinity-mini","name":"Trinity Mini","description":"Reasoning-tuned 26B MoE model with 3B active parameters for agents, tools, and multi-step workloads","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-01","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.045,"output":0.15}}}},"morph":{"id":"morph","env":["MORPH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.morphllm.com/v1","name":"Morph","doc":"https://docs.morphllm.com/api-reference/introduction","models":{"morph-v3-large":{"id":"morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"morph-v3-fast":{"id":"morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"auto":{"id":"auto","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.85,"output":1.55}}}},"aihubmix":{"id":"aihubmix","env":["AIHUBMIX_API_KEY"],"npm":"@aihubmix/ai-sdk-provider","name":"AIHubMix","doc":"https://docs.aihubmix.com","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"deepseek-v4-flash-0731-fast":{"id":"deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731 Fast","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.28,"output":1.4,"cache_read":0.07}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.7746,"output":3.0984,"cache_read":0.015492}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":7.999,"cache_read":0.32167}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.22,"output":1.32,"cache_read":0.044}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.155,"output":0.62,"cache_read":0.0031}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.6918,"output":2.0754,"cache_read":0.023058}},"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Doubao Seed 2.0 Lite 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.08,"output":0.51,"cache_read":0.01692,"input_audio":1.269,"tiers":[{"input":0.13,"output":0.76,"cache_read":0.02536,"input_audio":1.902,"tier":{"type":"context","size":32000}},{"input":0.25,"output":1.52,"cache_read":0.05072,"input_audio":3.804,"tier":{"type":"context","size":128000}}]}},"gemini-3.1-flash-lite-image":{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675,"cache_read":0.165}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.142,"output":0.284,"cache_read":0.0284}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675}},"coding-minimax-m2.7":{"id":"coding-minimax-m2.7","name":"Coding MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"coding-glm-5.1":{"id":"coding-glm-5.1","name":"Coding GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.22,"cache_read":0.013}},"claude-opus-4-7-think":{"id":"claude-opus-4-7-think","name":"Claude Opus 4.7 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.28,"output":1.69,"cache_read":0.0282,"cache_write":0.3525,"tiers":[{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41}}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Doubao Seed 2.0 Mini 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.03,"output":0.28,"cache_read":0.00564,"input_audio":0.423,"tiers":[{"input":0.06,"output":0.56,"cache_read":0.01128,"input_audio":0.846,"tier":{"type":"context","size":32000}},{"input":0.11,"output":1.13,"cache_read":0.02256,"input_audio":1.692,"tier":{"type":"context","size":128000}}]}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"doubao-seed-2-0-code-preview":{"id":"doubao-seed-2-0-code-preview","name":"Doubao Seed 2.0 Code Preview","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"xiaomi-mimo-v2.5-free":{"id":"xiaomi-mimo-v2.5-free","name":"Xiaomi MiMo-V2.5 (free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"coding-xiaomi-mimo-v2.5-pro":{"id":"coding-xiaomi-mimo-v2.5-pro","name":"Coding Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.2,"output":0.6,"cache_read":0.04,"tiers":[{"input":0.4,"output":1.2,"cache_read":0.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.4,"output":1.2,"cache_read":0.08}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.288,"output":1.152}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":3.9995,"cache_read":0.160835}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.5}},"alicloud-deepseek-v4-pro":{"id":"alicloud-deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.69,"output":3.38,"cache_read":0.13}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xiaomi-mimo-v2.5-pro-free":{"id":"xiaomi-mimo-v2.5-pro-free","name":"Xiaomi MiMo-V2.5-Pro (free)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.155,"output":0.62,"cache_read":0.0031}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.1562,"output":0.6248,"cache_read":0.03905}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.845,"output":2.535,"cache_read":0.04225}},"deep-deepseek-v4-pro":{"id":"deep-deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.478,"output":0.956,"cache_read":0.004302}},"deep-deepseek-v4-flash":{"id":"deep-deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepSeek)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.154,"output":0.308,"cache_read":0.0308}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":1.5}},"xiaomi-mimo-v2.5":{"id":"xiaomi-mimo-v2.5","name":"Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.088,"tiers":[{"input":0.88,"output":4.4,"cache_read":0.176,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.88,"output":4.4,"cache_read":0.176}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":64000},"cost":{"input":0.0282,"output":0.1128,"cache_read":0.00564,"cache_write":0.03525}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"alicloud-deepseek-v4-flash":{"id":"alicloud-deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3.8-omni-flash":{"id":"qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1126,"output":0.380025,"cache_read":0.014075,"cache_write":0.175937}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"interleaved":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"zai-glm-5.1":{"id":"zai-glm-5.1","name":"GLM-5.1 (Z.ai)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.845,"output":3.38,"cache_read":0.183112}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.11268,"output":0.39438,"cache_read":0.02817}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.499999,"cache_read":0.03}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":2,"output":6,"cache_read":0.5}},"claude-opus-4-8-think":{"id":"claude-opus-4-8-think","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675}},"coding-xiaomi-mimo-v2.5":{"id":"coding-xiaomi-mimo-v2.5","name":"Coding Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.08,"output":0.4,"cache_read":0.016,"tiers":[{"input":0.16,"output":0.8,"cache_read":0.032,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.16,"output":0.8,"cache_read":0.032}}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.17,"output":1.01,"cache_read":0.0169,"cache_write":0.21125,"tiers":[{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845}}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1126,"output":0.380025,"cache_read":0.014075,"cache_write":0.175937}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"coding-minimax-m2.7-free":{"id":"coding-minimax-m2.7-free","name":"Coding MiniMax M2.7 (Free)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0,"output":0}},"doubao-seed-2-0-pro":{"id":"doubao-seed-2-0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.27,"output":7.61,"cache_read":0.1268,"cache_write":1.585,"tiers":[{"input":2.11,"output":12.67,"cache_read":0.2112,"cache_write":2.64,"tier":{"type":"context","size":128000}}]}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"alicloud-glm-5.1":{"id":"alicloud-glm-5.1","name":"GLM-5.1 (Alibaba Cloud)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.84,"output":3.38,"cache_read":0.169,"cache_write":1.05625}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"claude-sonnet-4-6-think":{"id":"claude-sonnet-4-6-think","name":"Claude Sonnet 4.6 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.282,"output":1.128,"cache_read":0.0564,"cache_write":0.3525}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"hy3-preview":{"id":"hy3-preview","name":"Hy3 Preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.17,"output":0.566661,"cache_read":0.051}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM 5 Vision Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.7042,"output":3.09848,"cache_read":0.169008}},"ox-alpha":{"id":"ox-alpha","name":"Ox Alpha","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"xiaomi-mimo-v2.5-pro":{"id":"xiaomi-mimo-v2.5-pro","name":"Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.1,"output":3.3,"cache_read":0.22,"tiers":[{"input":2.2,"output":6.6,"cache_read":0.44,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.2,"output":6.6,"cache_read":0.44}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6-think":{"id":"claude-opus-4-6-think","name":"Claude Opus 4.6 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"coding-glm-5.1-free":{"id":"coding-glm-5.1-free","name":"Coding GLM 5.1 (free)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"coding-minimax-m2.7-highspeed":{"id":"coding-minimax-m2.7-highspeed","name":"Coding MiniMax M2.7 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"chutes":{"id":"chutes","env":["CHUTES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.chutes.ai/v1","name":"Chutes","doc":"https://llm.chutes.ai/v1/models","models":{"Nemotron-3-Nano-Omni-30B-TEE":{"id":"Nemotron-3-Nano-Omni-30B-TEE","name":"Nemotron 3 Nano Omni 30B TEE","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":0},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"deepseek-ai/DeepSeek-V4-Flash-0731-TEE":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731-TEE","name":"DeepSeek V4 Flash 0731 TEE","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.04399999999999999}},"deepseek-ai/DeepSeek-V3.2-TEE":{"id":"deepseek-ai/DeepSeek-V3.2-TEE","name":"DeepSeek V3.2 TEE","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":1,"cache_read":0.09999999999999998}},"google/gemma-4-31B-turbo-TEE":{"id":"google/gemma-4-31B-turbo-TEE","name":"gemma 4 31B turbo TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.12,"output":0.37,"cache_read":0.011999999999999997}},"zai-org/GLM-5.1-TEE":{"id":"zai-org/GLM-5.1-TEE","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":65535},"cost":{"input":0.98,"output":3.08,"cache_read":0.09799999999999998}},"zai-org/GLM-5.2-TEE":{"id":"zai-org/GLM-5.2-TEE","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":3.95,"cache_read":0.12499999999999997}},"Qwen/Qwen3.8-27B-TEE":{"id":"Qwen/Qwen3.8-27B-TEE","name":"Qwen3.8 27B TEE","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.24,"output":2.2,"cache_read":0.023999999999999994}},"Qwen/Qwen3.6-27B-TEE":{"id":"Qwen/Qwen3.6-27B-TEE","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2,"cache_read":0.029999999999999992}},"Qwen/Qwen3.5-397B-A17B-TEE":{"id":"Qwen/Qwen3.5-397B-A17B-TEE","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3,"cache_read":0.04499999999999999}},"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE","name":"Qwen3 235B A22B Thinking 2507 TEE","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2989,"output":1.1957,"cache_read":0.029889999999999993}},"Qwen/Qwen3-32B-TEE":{"id":"Qwen/Qwen3-32B-TEE","name":"Qwen3 32B TEE","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.104,"output":0.416,"cache_read":0.010399999999999998}},"unsloth/Mistral-Nemo-Instruct-2407-TEE":{"id":"unsloth/Mistral-Nemo-Instruct-2407-TEE","name":"Mistral Nemo Instruct 2407 TEE","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"moonshotai/Kimi-K3-TEE":{"id":"moonshotai/Kimi-K3-TEE","name":"Kimi K3 TEE","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":0.29999999999999993}},"moonshotai/Kimi-K2.6-TEE":{"id":"moonshotai/Kimi-K2.6-TEE","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65535},"cost":{"input":0.5,"output":2.85,"cache_read":0.04999999999999999}}}},"groq":{"id":"groq","env":["GROQ_API_KEY"],"npm":"@ai-sdk/groq","name":"Groq","doc":"https://console.groq.com/docs/models","models":{"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large V3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Llama 3.1 8B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.08}},"allam-2-7b":{"id":"allam-2-7b","name":"ALLaM-2-7b","description":"ALLaM-2-7b instruction tuned model by SDAIA","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.59,"output":0.79}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131042,"output":16384},"cost":{"input":0.8,"output":4}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.6,"output":3,"cache_read":0.3}},"groq/compound":{"id":"groq/compound","name":"Compound","description":"General-purpose chat model for instruction following, writing, and analysis","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"groq/compound-mini":{"id":"groq/compound-mini","name":"Compound Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"meta-llama/llama-prompt-guard-2-86m":{"id":"meta-llama/llama-prompt-guard-2-86m","name":"Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.04,"output":0.04}},"meta-llama/llama-prompt-guard-2-22m":{"id":"meta-llama/llama-prompt-guard-2-22m","name":"Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.03,"output":0.03}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"Safety GPT OSS 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta","cost":{"input":0.075,"output":0.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-10-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"canopylabs/orpheus-v1-english":{"id":"canopylabs/orpheus-v1-english","name":"Canopy Labs Orpheus V1 English","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"},"canopylabs/orpheus-arabic-saudi":{"id":"canopylabs/orpheus-arabic-saudi","name":"Canopy Labs Orpheus Arabic Saudi","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"}}},"zai-coding-plan":{"id":"zai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/coding/paas/v4","name":"Z.AI Coding Plan","doc":"https://docs.z.ai/devpack/overview","models":{"glm-5.2-highspeed":{"id":"glm-5.2-highspeed","name":"GLM-5.2 Highspeed","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"volcengine":{"id":"volcengine","env":["ARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/v3","name":"Volcengine Ark","doc":"https://www.volcengine.com/docs/82379/1330310","models":{"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.08906,"output":0.53436,"cache_read":0.01781,"tiers":[{"input":0.13359,"output":0.80154,"cache_read":0.02672,"tier":{"type":"context","size":32000}},{"input":0.26718,"output":1.60308,"cache_read":0.05344,"tier":{"type":"context","size":128000}}]}},"doubao-seed-character-260628":{"id":"doubao-seed-character-260628","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.11875,"output":0.29687,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":0.8906,"cache_read":0.02375,"tier":{"type":"context","size":32000}}]}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.02969,"output":0.29687,"cache_read":0.00594,"tiers":[{"input":0.05937,"output":0.59374,"cache_read":0.01187,"tier":{"type":"context","size":32000}},{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-1-pro-260628":{"id":"doubao-seed-2-1-pro-260628","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-8-251228":{"id":"doubao-seed-1-8-251228","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-flash-250828":{"id":"doubao-seed-1-6-flash-250828","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.02227,"output":0.22265,"cache_read":0.00445,"tiers":[{"input":0.04453,"output":0.4453,"cache_read":0.00445,"tier":{"type":"context","size":32000}},{"input":0.08906,"output":0.8906,"cache_read":0.00445,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-251015":{"id":"doubao-seed-1-6-251015","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-pro-ga-260813":{"id":"deepseek-v4-pro-ga-260813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.3359,"output":4.00771,"cache_read":0.04453}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"glm-5-2-260617":{"id":"glm-5-2-260617","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.18747,"output":4.15615,"cache_read":0.29687}},"doubao-seed-2-1-turbo-260628":{"id":"doubao-seed-2-1-turbo-260628","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.4453,"output":2.22651,"cache_read":0.08906}},"glm-5-3-flash-260828":{"id":"glm-5-3-flash-260828","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11875,"output":0.41563,"cache_read":0.03414}},"deepseek-v4-flash-ga-260731":{"id":"deepseek-v4-flash-ga-260731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4453,"output":1.3359,"cache_read":0.01484}}}},"sensenova":{"id":"sensenova","env":["SENSENOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token.sensenova.cn/v1","name":"SenseNova (China)","doc":"https://platform.sensenova.cn/docs","models":{"sensenova-6.8-flash-lite":{"id":"sensenova-6.8-flash-lite","name":"SenseNova 6.8 Flash Lite","description":"SenseNova lightweight multimodal agent model for real-world complex tasks, data analysis, and complex information presentation","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}}}},"orcarouter":{"id":"orcarouter","env":["ORCAROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.orcarouter.ai/v1","name":"OrcaRouter","doc":"https://docs.orcarouter.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.563}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.086,"output":0.688}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.33,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.057,"output":0.459}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.115,"output":0.917}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.35,"output":1.42,"cache_read":0.071}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.115,"output":0.688,"reasoning":2.4}},"orcarouter/free":{"id":"orcarouter/free","name":"OrcaRouter Free","description":"Built-in router over the free tier that scores each request's difficulty and sends light work to the smaller free model and harder work to the stronger one. Priced at zero and never falls back to a paid model.","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0,"output":0}},"orcarouter/fusion-mini":{"id":"orcarouter/fusion-mini","name":"OrcaRouter Fusion Mini","description":"Leaner two-model Fusion panel that runs Claude Opus 4.8 and GPT-5.5 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/fusion":{"id":"orcarouter/fusion","name":"OrcaRouter Fusion","description":"Curated fan-out router that runs Claude Opus 4.8, GPT-5.5 and Gemini 3.1 Pro in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/fusion-flash":{"id":"orcarouter/fusion-flash","name":"OrcaRouter Fusion Flash","description":"Budget Fusion panel that runs Gemini 3.5 Flash, MiniMax M2.7 and GLM 5.1 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Cost-sensitive fan-out over a 200K window.","family":"model-router","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"orcarouter/auto":{"id":"orcarouter/auto","name":"OrcaRouter Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2026-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":10}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.33,"cache_read":0.0075}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333,"input_audio":3}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-robotics-er-1.6-preview":{"id":"google/gemini-robotics-er-1.6-preview","name":"Gemini Robotics-ER 1.6 Preview","description":"Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":5}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38,"cache_read":0.02}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"grok/grok-4.3":{"id":"grok/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok/grok-4.5":{"id":"grok/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"grok/grok-4.6":{"id":"grok/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-flash-free":{"id":"deepseek/deepseek-v4-flash-free","name":"DeepSeek V4 Flash (free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-reasoner":{"id":"deepseek/deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.028}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":100000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":100000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.17}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"kimi/kimi-k2.6":{"id":"kimi/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi/kimi-k2.7-code":{"id":"kimi/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi/kimi-k3":{"id":"kimi/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.3,"output":16.5,"cache_read":0.33}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1,"cache_write":0}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.18,"output":0.59,"cache_read":0.059}},"tencent/hy3-free":{"id":"tencent/hy3-free","name":"Hy3 (free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.075,"output":0.25}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.26,"cache_write":0}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"z-ai/glm-5.3-flash-free":{"id":"z-ai/glm-5.3-flash-free","name":"GLM-5.3-Flash (free)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}}}},"routing-run":{"id":"routing-run","env":["ROUTING_RUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.routing.run/v1","name":"routing.run","doc":"https://docs.routing.run/api-reference/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.16,"output":0.48}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.112,"output":0.224}},"kimi-k2.6-nitro":{"id":"kimi-k2.6-nitro","name":"Kimi K2.6 Nitro","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.1,"output":0.1}},"glm-5.2-nitro":{"id":"glm-5.2-nitro","name":"GLM 5.2 Nitro","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":0.7,"output":4.2}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":5,"output":25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.348,"output":0.696}},"kimi-k2.7-code-nitro":{"id":"kimi-k2.7-code-nitro","name":"Kimi K2.7 Code Nitro","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":1.5,"output":9}}}},"llmtech":{"id":"llmtech","env":["LLMTECH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmtech.eu/v1","name":"LLM Tech","doc":"https://llmtech.eu/models/qwen3.8-27b","models":{"nvidia/Qwen3.8-27B-NVFP4":{"id":"nvidia/Qwen3.8-27B-NVFP4","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2.09,"cache_read":0.04}}}},"sap-ai-core":{"id":"sap-ai-core","env":["AICORE_SERVICE_KEY"],"npm":"@jerome-benoit/sap-ai-provider-v2","name":"SAP AI Core","doc":"https://help.sap.com/docs/sap-ai-core","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.08,"output":0.26}},"anthropic--claude-4.5-sonnet":{"id":"anthropic--claude-4.5-sonnet","name":"anthropic--claude-4.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-medium":{"id":"mistralai--mistral-medium","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"gpt-5.6-sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"anthropic--claude-4.5-haiku":{"id":"anthropic--claude-4.5-haiku","name":"anthropic--claude-4.5-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"cohere--command-a-reasoning":{"id":"cohere--command-a-reasoning","name":"cohere--command-a-reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.63,"output":5.05}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"anthropic--claude-4.8-opus":{"id":"anthropic--claude-4.8-opus","name":"anthropic--claude-4.8-opus","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"gemini-3.1-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"gemini-3.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"gpt-5.6-luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"amazon--titan-embed-text":{"id":"amazon--titan-embed-text","name":"amazon--titan-embed-text","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-04-30","last_updated":"2024-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.14,"output":0}},"anthropic--claude-4.5-opus":{"id":"anthropic--claude-4.5-opus","name":"anthropic--claude-4.5-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic--claude-3.5-sonnet":{"id":"anthropic--claude-3.5-sonnet","name":"anthropic--claude-3.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-medium-instruct":{"id":"mistralai--mistral-medium-instruct","name":"mistralai--mistral-medium-instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.36,"output":1.22}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.32}},"anthropic--claude-4.6-opus":{"id":"anthropic--claude-4.6-opus","name":"anthropic--claude-4.6-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"sonar":{"id":"sonar","name":"sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"anthropic--claude-4-sonnet":{"id":"anthropic--claude-4-sonnet","name":"anthropic--claude-4-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-small":{"id":"mistralai--mistral-small","name":"mistralai--mistral-small","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.28}},"amazon--nova-pro":{"id":"amazon--nova-pro","name":"amazon--nova-pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.56,"output":2.13}},"anthropic--claude-3-opus":{"id":"anthropic--claude-3-opus","name":"anthropic--claude-3-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"nvidia--llama-3.2-nv-embedqa-1b":{"id":"nvidia--llama-3.2-nv-embedqa-1b","name":"nvidia--llama-3.2-nv-embedqa-1b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.07,"output":0}},"anthropic--claude-4.7-opus":{"id":"anthropic--claude-4.7-opus","name":"anthropic--claude-4.7-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":3072}},"sap-abap-1":{"id":"sap-abap-1","name":"sap-abap-1","description":"SAP-hosted model for ABAP code generation and enterprise development tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.48,"output":1.7}},"amazon--nova-lite":{"id":"amazon--nova-lite","name":"amazon--nova-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.3,"output":2.37}},"anthropic--claude-3-haiku":{"id":"anthropic--claude-3-haiku","name":"anthropic--claude-3-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"sonar-pro":{"id":"sonar-pro","name":"sonar-pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"amazon--nova-micro":{"id":"amazon--nova-micro","name":"amazon--nova-micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.1}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"sonar-deep-research":{"id":"sonar-deep-research","name":"sonar-deep-research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.09,"output":0}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-25","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"gpt-5.6-terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":9.44,"cache_read":0.12}},"anthropic--claude-4.6-sonnet":{"id":"anthropic--claude-4.6-sonnet","name":"anthropic--claude-4.6-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-17","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"anthropic--claude-4-opus":{"id":"anthropic--claude-4-opus","name":"anthropic--claude-4-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5.5":{"id":"gpt-5.5","name":"gpt-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"anthropic--claude-3-sonnet":{"id":"anthropic--claude-3-sonnet","name":"anthropic--claude-3-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-04","last_updated":"2024-03-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-embedding":{"id":"gemini-embedding","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1}},"anthropic--claude-3.7-sonnet":{"id":"anthropic--claude-3.7-sonnet","name":"anthropic--claude-3.7-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}}}},"alibaba-coding-plan-cn":{"id":"alibaba-coding-plan-cn","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan (China)","doc":"https://help.aliyun.com/zh/model-studio/coding-plan","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"azure-cognitive-services":{"id":"azure-cognitive-services","env":["AZURE_COGNITIVE_SERVICES_RESOURCE_NAME","AZURE_COGNITIVE_SERVICES_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure Cognitive Services","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":15,"output":120}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}}}},"regolo-ai":{"id":"regolo-ai","env":["REGOLO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.regolo.ai/v1","name":"Regolo AI","doc":"https://docs.regolo.ai/","models":{"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":120000},"cost":{"input":0.58,"output":2.42}},"faster-whisper-large-v3":{"id":"faster-whisper-large-v3","name":"Faster Whisper Large v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":100000},"cost":{"input":0.46,"output":2.42}},"brick-complexity-pro":{"id":"brick-complexity-pro","name":"Brick Complexity Pro","description":"Complexity classifier that powers the Brick semantic router by extracting query difficulty","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"cost":{"input":0.12,"output":0.46}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":30000},"cost":{"input":0.46,"output":2.42}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT-OSS-20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.4,"output":1.8}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3,"output":1.2}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.1,"output":0.1}},"qwen3-reranker-4b":{"id":"qwen3-reranker-4b","name":"Qwen3-Reranker-4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.12,"output":0.12}},"glm5.2":{"id":"glm5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":96000,"output":96000},"cost":{"input":2.31,"output":6}},"brick-v1-beta":{"id":"brick-v1-beta","name":"Brick v1 Beta","description":"Semantic router by Regolo.ai that directs each request to the most suitable model, optimizing costs and performance","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"status":"beta","cost":{"input":0,"output":0}},"qwen-image":{"id":"qwen-image","name":"Qwen-Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS-120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1,"output":4.2}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":4000},"cost":{"input":0,"output":0}},"mistral-small-4-119b":{"id":"mistral-small-4-119b","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.75,"output":3}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.7}},"qwen3.5-122b":{"id":"qwen3.5-122b","name":"Qwen3.5-122B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.9,"output":3.6}}}},"kenari":{"id":"kenari","env":["KENARI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://kenari.id/v1","name":"Kenari","doc":"https://kenari.id/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-1":{"id":"glm-5-1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-2-5-flash":{"id":"gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"kimi-k2-7-code:free":{"id":"kimi-k2-7-code:free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash (Free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"mistral-medium-3-5:free":{"id":"mistral-medium-3-5:free","name":"Mistral Medium 3.5 (Free)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"glm-5-3":{"id":"glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"step-3-7-flash:free":{"id":"step-3-7-flash:free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"grok-imagine-image-2-0":{"id":"grok-imagine-image-2-0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":8000,"output":0},"cost":{"input":0,"output":0}},"kimi-k2-6:free":{"id":"kimi-k2-6:free","name":"Kimi K2.6 (Free)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"hy3:free":{"id":"hy3:free","name":"Hy3 (Free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"mistral-large:free":{"id":"mistral-large:free","name":"Mistral Large (Free)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"mimo-v2-5:free":{"id":"mimo-v2-5:free","name":"MiMo-V2.5 (Free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b:free":{"id":"nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super 120B A12B (Free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0,"output":0}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"glm-4-7-flash:free":{"id":"glm-4-7-flash:free","name":"GLM-4.7-Flash (Free)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":272000,"output":16384},"cost":{"input":0,"output":0}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gemini-2-5-flash-lite":{"id":"gemini-2-5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemini-3-1-flash-tts":{"id":"gemini-3-1-flash-tts","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0,"output":0}},"nemotron-3-nano-30b-a3b":{"id":"nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}}}},"the-grid-ai":{"id":"the-grid-ai","env":["THEGRID_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.thegrid.ai/v1","name":"The Grid AI","doc":"https://thegrid.ai/docs","models":{"agent-prime":{"id":"agent-prime","name":"Agent Prime","description":"Reliable models for dependable agentic applications, multi-step tool use, and reasoning workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"text-standard":{"id":"text-standard","name":"Text Standard","description":"Price-optimized models with low-latency, high-throughput and shorter maximum outputs. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000}},"agent-max":{"id":"agent-max","name":"Agent Max","description":"Frontier models for autonomous research, deep multi-step tool chains, and complex long-horizon tasks. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"text-max":{"id":"text-max","name":"Text Max","description":"Frontier models for deep reasoning, long context, and complex workflows. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000}},"code-max":{"id":"code-max","name":"Code Max","description":"Frontier models for complex research, architectural decisions, debugging, and multi-file development. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"code-standard":{"id":"code-standard","name":"Code Standard","description":"Price-optimized models for rapid autocomplete, linting, high-frequency suggestions, and batch edits. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"},"code-prime":{"id":"code-prime","name":"Code Prime","description":"Reliable models for everyday software tasks, code completion, review, and standard debugging. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"text-prime":{"id":"text-prime","name":"Text Prime","description":"Reliable models for everyday text generation, editing, and analysis across diverse workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000}},"agent-standard":{"id":"agent-standard","name":"Agent Standard","description":"Price-optimized models for fast tool calls, simple agent loops, high-throughput automation, and orchestration. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"}}},"google-vertex":{"id":"google-vertex","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex","name":"Vertex","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/models","models":{"gemini-2.5-flash-tts":{"id":"gemini-2.5-flash-tts","name":"Gemini 2.5 Flash TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-pro-tts":{"id":"gemini-2.5-pro-tts","name":"Gemini 2.5 Pro TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":1,"output":20}},"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":120,"cache_read":0.2}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60,"cache_read":0.05}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen/qwen3-235b-a22b-instruct-2507-maas":{"id":"qwen/qwen3-235b-a22b-instruct-2507-maas","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.22,"output":0.88}},"deepseek-ai/deepseek-v3.1-maas":{"id":"deepseek-ai/deepseek-v3.1-maas","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":1.7,"cache_read":0.06}},"deepseek-ai/deepseek-v3.2-maas":{"id":"deepseek-ai/deepseek-v3.2-maas","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-17","last_updated":"2026-04-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"zai-org/glm-5.2-maas":{"id":"zai-org/glm-5.2-maas","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai-org/glm-5-maas":{"id":"zai-org/glm-5-maas","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"zai-org/glm-4.7-maas":{"id":"zai-org/glm-4.7-maas","name":"GLM-4.7","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-06","last_updated":"2026-01-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.2,"cache_read":0.06}},"meta/llama-4-maverick-17b-128e-instruct-maas":{"id":"meta/llama-4-maverick-17b-128e-instruct-maas","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":8192},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.35,"output":1.15}},"meta/llama-3.3-70b-instruct-maas":{"id":"meta/llama-3.3-70b-instruct-maas","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.72,"output":0.72}},"openai/gpt-oss-120b-maas":{"id":"openai/gpt-oss-120b-maas","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.09,"output":0.36}},"openai/gpt-oss-20b-maas":{"id":"openai/gpt-oss-20b-maas","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.07,"output":0.25,"cache_read":0.007}},"moonshotai/kimi-k2-thinking-maas":{"id":"moonshotai/kimi-k2-thinking-maas","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"xai/grok-4.20-reasoning":{"id":"xai/grok-4.20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":30000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-non-reasoning":{"id":"xai/grok-4.20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast","description":"Fast Grok model for responsive chat, tool-assisted work, and low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":500000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}}}},"infer":{"id":"infer","env":["INFER_API_KEY"],"npm":"@ai-sdk/openai","api":"https://infer.flow7.org/v1","name":"Infer by Flow7","doc":"https://infer.flow7.org/opencode","models":{"infer/gpt-5.6-sol:official":{"id":"infer/gpt-5.6-sol:official","name":"GPT-5.6 Sol (Official API)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":271999,"output":128000},"provider":{"shape":"responses"},"cost":{"input":2.5,"output":12.5,"cache_read":0.25,"cache_write":3.125}},"infer/gpt-6-astra:official":{"id":"infer/gpt-6-astra:official","name":"GPT-6 Astra (Official API)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":271999,"output":128000},"provider":{"shape":"responses"},"cost":{"input":12.5,"output":62.5,"cache_read":1.25,"cache_write":15.625}}}},"stepfun-ai":{"id":"stepfun-ai","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/v1","name":"StepFun (Global)","doc":"https://platform.stepfun.ai/docs/en/overview/concept","models":{"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}}}},"pendra":{"id":"pendra","env":["PENDRA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pendra.ai/api/v1","name":"Pendra","doc":"https://pendra.ai/docs/integrations/opencode","models":{"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3-coder:30b":{"id":"qwen3-coder:30b","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.6:27b":{"id":"qwen3.6:27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"llama3.3:70b":{"id":"llama3.3:70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}}}},"above":{"id":"above","env":["ABOVE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.above.dev/v1","name":"above.dev","doc":"https://above.dev/docs","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision (Exp)","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.242,"output":0.726,"reasoning":0.726,"cache_read":0.0077}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.54,"output":4.84,"cache_read":0.154}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.165,"output":0.66,"reasoning":0.66,"cache_read":0.0033}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.31,"output":7.26,"cache_read":0.231}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.165,"output":0.55,"cache_read":0.0319}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen 3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.2,"output":6.6,"cache_read":0.275}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.726,"output":2.178,"reasoning":2.178,"cache_read":0.0242}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.5077,"output":1.0154,"cache_read":0.0042}}}},"scaleway":{"id":"scaleway","env":["SCALEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scaleway.ai/v1","name":"Scaleway","doc":"https://www.scaleway.com/en/docs/generative-apis/","models":{"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.25,"output":0.5}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.468,"output":0.936,"reasoning":0.936,"cache_read":0.0936}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.8,"output":5.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.8}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.6,"output":3.6}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.1,"output":0}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2026-03-17","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":8192},"cost":{"input":0.003,"output":0}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":0.25,"output":1.5}},"bge-multilingual-gemma2":{"id":"bge-multilingual-gemma2","name":"BGE Multilingual Gemma2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-26","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.1,"output":0}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5 128B","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.5,"output":7.5}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2 24B Instruct (2506)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.35}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":16384},"cost":{"input":0.75,"output":2.25,"reasoning":8.4}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.6}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B 2409","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-25","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":0.9,"output":0.9}}}},"alibaba-cn":{"id":"alibaba-cn","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope.aliyuncs.com/compatible-mode/v1","name":"Alibaba (China)","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":1.721}},"deepseek-r1-distill-qwen-7b":{"id":"deepseek-r1-distill-qwen-7b","name":"DeepSeek R1 Distill Qwen 7B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.072,"output":0.144}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2026-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0.574,"output":2.296,"tiers":[{"input":0.861,"output":3.444,"tier":{"type":"context","size":32000}},{"input":1.435,"output":5.74,"tier":{"type":"context","size":128000}},{"input":2.87,"output":28.7,"tier":{"type":"context","size":256000}}]}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":1.434}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.287,"output":1.147}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.087,"output":0.345,"input_audio":5.448}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.101,"output":0.28}},"deepseek-v3-1":{"id":"deepseek-v3-1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.574,"output":1.721}},"qwen-deep-research":{"id":"qwen-deep-research","name":"Qwen Deep Research","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":7.742,"output":23.367}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.144,"output":0.574}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.574,"reasoning":1.434}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.345,"output":1.377}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"moonshot-kimi-k2-instruct":{"id":"moonshot-kimi-k2-instruct","name":"Moonshot Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":2.294}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.115,"output":0.287}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Moonshot Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.929,"output":3.858}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.216}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1,"output":3.851,"cache_read":0.275,"cache_write":0}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.044,"output":0.087,"reasoning":0.431}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":0.431,"reasoning":1.076}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.043,"output":0.072}},"tongyi-intent-detect-v3":{"id":"tongyi-intent-detect-v3","name":"Tongyi Intent Detect V3","description":"General-purpose chat model for instruction following, writing, and analysis","family":"yi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1024},"cost":{"input":0.058,"output":0.144}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Moonshot Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.574,"output":2.294}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwen3.5-flash":{"id":"qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-23","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.172,"output":1.033,"reasoning":1.033,"tiers":[{"input":0.689,"output":4.133,"reasoning":4.133,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.689,"output":4.133,"reasoning":4.133}}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.143353,"output":1.433525,"reasoning":4.300576}},"deepseek-v3-2-exp":{"id":"deepseek-v3-2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.287,"output":0.431}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.29754,"output":1.19015,"cache_read":0.01488}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.144}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.216,"output":0.861,"tiers":[{"input":0.323,"output":1.291,"tier":{"type":"context","size":32000}},{"input":0.538,"output":2.151,"tier":{"type":"context","size":128000}}]}},"qwen2-5-math-7b-instruct":{"id":"qwen2-5-math-7b-instruct","name":"Qwen2.5-Math 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.144,"output":0.287}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032,"reasoning":1.032,"tiers":[{"input":0.43,"output":2.58,"reasoning":2.58,"tier":{"type":"context","size":128000}}]}},"qwq-32b":{"id":"qwq-32b","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.032,"output":0.032}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"deepseek-r1-distill-qwen-1-5b":{"id":"deepseek-r1-distill-qwen-1-5b","name":"DeepSeek R1 Distill Qwen 1.5B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.02962,"output":0.1185,"cache_read":0.002962,"cache_write":0.03703,"tiers":[{"input":0.08887,"output":0.35549,"cache_read":0.008887,"cache_write":0.11109,"tier":{"type":"context","size":32000}},{"input":0.17774,"output":0.71098,"cache_read":0.017774,"cache_write":0.22218,"tier":{"type":"context","size":256000}}]}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":2.827,"output":14.133,"cache_read":0.283}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.294,"output":6.881}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"deepseek-r1-distill-qwen-32b":{"id":"deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2026-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.291,"output":7.749,"tiers":[{"input":2.153,"output":12.915,"tier":{"type":"context","size":128000}}]}},"qwen2-5-coder-32b-instruct":{"id":"qwen2-5-coder-32b-instruct","name":"Qwen2.5-Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"qwen2-5-math-72b-instruct":{"id":"qwen2-5-math-72b-instruct","name":"Qwen2.5-Math 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.115,"output":0.287,"reasoning":1.147,"cache_read":0.012,"cache_write":0.144}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.717}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11875,"output":0.40073,"cache_read":0.01187,"cache_write":0.14844}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.431}},"qwen-math-turbo":{"id":"qwen-math-turbo","name":"Qwen Math Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.287,"output":0.861}},"qwen-plus-character":{"id":"qwen-plus-character","name":"Qwen Plus Character","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.115,"output":0.287}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245800,"output":65536},"cost":{"input":1.32,"output":7.9,"cache_read":0.132}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58,"tiers":[{"input":0.86,"output":3.154,"tier":{"type":"context","size":32000}}]}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.77744,"output":5.33231,"cache_read":0.22218,"cache_write":2.22179}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Moonshot Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.574,"output":2.411}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.287,"reasoning":0.717}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.17,"tiers":[{"input":1.1,"output":3.851,"tier":{"type":"context","size":32000}}]}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":128000}}]}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.861,"output":3.441,"tiers":[{"input":1.291,"output":5.161,"tier":{"type":"context","size":32000}},{"input":2.151,"output":8.602,"tier":{"type":"context","size":128000}}]}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"qwen2-5-coder-7b-instruct":{"id":"qwen2-5-coder-7b-instruct","name":"Qwen2.5-Coder 7B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.287}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1,"output":3.851,"cache_read":0.275,"cache_write":0}},"qwen-long":{"id":"qwen-long","name":"Qwen Long","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"output":8192},"cost":{"input":0.072,"output":0.287}},"deepseek-r1-distill-llama-8b":{"id":"deepseek-r1-distill-llama-8b","name":"DeepSeek R1 Distill Llama 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"deepseek-r1-distill-qwen-14b":{"id":"deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.144,"output":0.431}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.286705,"output":1.14682,"reasoning":2.867051}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.287,"output":1.722,"reasoning":1.722,"tiers":[{"input":1.148,"output":6.888,"reasoning":6.888,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.148,"output":6.888,"reasoning":6.888}}},"qwen-math-plus":{"id":"qwen-math-plus","name":"Qwen Math Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-08-16","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen-doc-turbo":{"id":"qwen-doc-turbo","name":"Qwen Doc Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.087,"output":0.144}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.259,"output":0.775}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.147,"output":4.588}},"MiniMax/MiniMax-M2.7":{"id":"MiniMax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"siliconflow/deepseek-r1-0528":{"id":"siliconflow/deepseek-r1-0528","name":"siliconflow/deepseek-r1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.18}},"siliconflow/deepseek-v3.2":{"id":"siliconflow/deepseek-v3.2","name":"siliconflow/deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.42}},"siliconflow/deepseek-v3.1-terminus":{"id":"siliconflow/deepseek-v3.1-terminus","name":"siliconflow/deepseek-v3.1-terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":1}},"siliconflow/deepseek-v3-0324":{"id":"siliconflow/deepseek-v3-0324","name":"siliconflow/deepseek-v3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":1}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"kimi/kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}}}},"poe":{"id":"poe","env":["POE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.poe.com/v1","name":"Poe","doc":"https://creator.poe.com/docs/external-applications/openai-compatible-api","models":{"poetools/claude-code":{"id":"poetools/claude-code","name":"claude-code","description":"Claude model for careful reasoning, writing, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-27","last_updated":"2025-11-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"elevenlabs/elevenlabs-v2.5-turbo":{"id":"elevenlabs/elevenlabs-v2.5-turbo","name":"ElevenLabs-v2.5-Turbo","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-28","last_updated":"2024-10-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-v3":{"id":"elevenlabs/elevenlabs-v3","name":"ElevenLabs-v3","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-music":{"id":"elevenlabs/elevenlabs-music","name":"ElevenLabs-Music","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-29","last_updated":"2025-08-29","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":2000,"output":0}},"stabilityai/stablediffusionxl":{"id":"stabilityai/stablediffusionxl","name":"StableDiffusionXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-07-09","last_updated":"2023-07-09","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":200,"output":0}},"trytako/tako":{"id":"trytako/tako","name":"Tako","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"tako","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":0}},"ideogramai/ideogram-v2a":{"id":"ideogramai/ideogram-v2a","name":"Ideogram-v2a","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2a-turbo":{"id":"ideogramai/ideogram-v2a-turbo","name":"Ideogram-v2a-Turbo","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2":{"id":"ideogramai/ideogram-v2","name":"Ideogram-v2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-21","last_updated":"2024-08-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram":{"id":"ideogramai/ideogram","name":"Ideogram","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-04-03","last_updated":"2024-04-03","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude-Opus-4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.2929,"output":21.4646}},"anthropic/claude-sonnet-3.5":{"id":"anthropic/claude-sonnet-3.5","name":"Claude-Sonnet-3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-06-05","last_updated":"2024-06-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude-Opus-4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.4}},"anthropic/claude-sonnet-3.7":{"id":"anthropic/claude-sonnet-3.7","name":"Claude-Sonnet-3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude-Opus-4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":32000},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"anthropic/claude-haiku-3":{"id":"anthropic/claude-haiku-3","name":"Claude-Haiku-3","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-09","last_updated":"2024-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.21,"output":1.1,"cache_read":0.021,"cache_write":0.26}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude-Sonnet-4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-haiku-3.5":{"id":"anthropic/claude-haiku-3.5","name":"Claude-Haiku-3.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.68,"output":3.4,"cache_read":0.068,"cache_write":0.85}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude-Haiku-4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":64000},"cost":{"input":0.85,"output":4.3,"cache_read":0.085,"cache_write":1.1}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude-Opus-4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude-Opus-4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192512,"output":28672},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude-Sonnet-4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":32768},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-sonnet-3.5-june":{"id":"anthropic/claude-sonnet-3.5-june","name":"Claude-Sonnet-3.5-June","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude-Opus-4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-11-21","last_updated":"2025-11-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":64000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude-Sonnet-4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":64000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"google/nano-banana-pro":{"id":"google/nano-banana-pro","name":"Nano-Banana-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-3.1-pro":{"id":"google/gemini-3.1-pro","name":"Gemini-3.1-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-deep-research":{"id":"google/gemini-deep-research","name":"gemini-deep-research","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":0},"status":"deprecated","cost":{"input":1.6,"output":9.6}},"google/gemini-2.0-flash":{"id":"google/gemini-2.0-flash","name":"Gemini-2.0-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.1,"output":0.42}},"google/veo-3.1-fast":{"id":"google/veo-3.1-fast","name":"Veo-3.1-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/nano-banana":{"id":"google/nano-banana","name":"Nano-Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/imagen-4":{"id":"google/imagen-4","name":"Imagen-4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini-2.5-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-06-19","last_updated":"2025-06-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":64000},"cost":{"input":0.07,"output":0.28}},"google/imagen-3-fast":{"id":"google/imagen-3-fast","name":"Imagen-3-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-17","last_updated":"2024-10-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.0-flash-lite":{"id":"google/gemini-2.0-flash-lite","name":"Gemini-2.0-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.052,"output":0.21}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini-3.1-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"google/veo-3.1":{"id":"google/veo-3.1","name":"Veo-3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo-3-fast":{"id":"google/veo-3-fast","name":"Veo-3-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4-fast":{"id":"google/imagen-4-fast","name":"Imagen-4-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini-3.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5152,"output":9.0909,"cache_read":0.1515}},"google/veo-3":{"id":"google/veo-3","name":"Veo-3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3-pro":{"id":"google/gemini-3-pro","name":"Gemini-3-Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":1.6,"output":9.6,"cache_read":0.16}},"google/lyria":{"id":"google/lyria","name":"Lyria","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-04","last_updated":"2025-06-04","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemma-4-31b":{"id":"google/gemma-4-31b","name":"Gemma-4-31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"google/imagen-4-ultra":{"id":"google/imagen-4-ultra","name":"Imagen-4-Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-24","last_updated":"2025-05-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo-2":{"id":"google/veo-2","name":"Veo-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini-2.5-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":32768}],"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.87,"output":7,"cache_read":0.087}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini-3-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini-2.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-04-26","last_updated":"2025-04-26","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/imagen-3":{"id":"google/imagen-3","name":"Imagen-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"glm-4.7","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"status":"deprecated"},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"minimax-m2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-26","last_updated":"2025-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"glm-4.6v","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":32768}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-05-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.96,"output":4.04,"cache_read":0.16}},"novita/kimi-k2-thinking":{"id":"novita/kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":0}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":0},"cost":{"input":0.27,"output":0.4,"cache_read":0.13}},"novita/glm-4.7-n":{"id":"novita/glm-4.7-n","name":"glm-4.7-n","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/kimi-k2.5":{"id":"novita/kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"novita/glm-4.7-flash":{"id":"novita/glm-4.7-flash","name":"glm-4.7-flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65500}},"fireworks-ai/kimi-k2.5-fw":{"id":"fireworks-ai/kimi-k2.5-fw","name":"Kimi-K2.5-FW","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":245760,"output":16384},"cost":{"input":0,"output":0}},"lumalabs/ray2":{"id":"lumalabs/ray2","name":"Ray2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ray","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":5000,"output":0}},"empiriolabs/deepseek-v4-flash-el":{"id":"empiriolabs/deepseek-v4-flash-el","name":"DeepSeek-V4-Flash-EL","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.14,"output":0.28}},"empiriolabs/deepseek-v4-pro-el":{"id":"empiriolabs/deepseek-v4-pro-el","name":"DeepSeek-V4-Pro-EL","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":1.67,"output":3.33}},"topazlabs-co/topazlabs":{"id":"topazlabs-co/topazlabs","name":"TopazLabs","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"topazlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":204,"output":0}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36,"cache_read":0.0045}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09,"output":0.36,"cache_read":0.022}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":14,"output":110}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3-mini-high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/chatgpt-4o-latest":{"id":"openai/chatgpt-4o-latest","name":"ChatGPT-4o-Latest","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"status":"deprecated","cost":{"input":4.5,"output":14}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4-classic":{"id":"openai/gpt-4-classic","name":"GPT-4-Classic","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-25","last_updated":"2024-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/o3-deep-research":{"id":"openai/o3-deep-research","name":"o3-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":9,"output":36,"cache_read":2.2}},"openai/sora-2":{"id":"openai/sora-2","name":"Sora-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.3-instant":{"id":"openai/gpt-5.3-instant","name":"GPT-5.3-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":19,"output":150}},"openai/gpt-5.3-codex-spark":{"id":"openai/gpt-5.3-codex-spark","name":"GPT-5.3-Codex-Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.36,"output":1.4,"cache_read":0.09}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":14,"cache_read":0.22}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":9,"output":27}},"openai/gpt-3.5-turbo-raw":{"id":"openai/gpt-3.5-turbo-raw","name":"GPT-3.5-Turbo-Raw","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":4524,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-5.2-instant":{"id":"openai/gpt-5.2-instant","name":"GPT-5.2-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/dall-e-3":{"id":"openai/dall-e-3","name":"DALL-E-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"dall-e","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":800,"output":0}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1-Codex-Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/o4-mini-deep-research":{"id":"openai/o4-mini-deep-research","name":"o4-mini-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":14,"output":54}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":124096,"output":4096},"cost":{"input":0.14,"output":0.54,"cache_read":0.068}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":140,"output":540}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT-Image-1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4-Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.18,"output":1.1,"cache_read":0.018}},"openai/gpt-4o-search":{"id":"openai/gpt-4o-search","name":"GPT-4o-Search","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5-Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":27.2727,"output":163.6364}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT-Image-1-Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-aug":{"id":"openai/gpt-4o-aug","name":"GPT-4o-Aug","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-21","last_updated":"2024-11-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9,"cache_read":1.1}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-12","last_updated":"2026-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.68,"output":4,"cache_read":0.068}},"openai/gpt-5.1-instant":{"id":"openai/gpt-5.1-instant","name":"GPT-5.1-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5.0505,"output":32.3232,"cache_read":1.2626}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-5-chat":{"id":"openai/gpt-5-chat","name":"GPT-5-Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":27,"output":160}},"openai/sora-2-pro":{"id":"openai/sora-2-pro","name":"Sora-2-Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-mini-search":{"id":"openai/gpt-4o-mini-search","name":"GPT-4o-mini-Search","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.14,"output":0.54}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5-Turbo-Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-20","last_updated":"2023-09-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":3500,"output":1024},"cost":{"input":1.4,"output":1.8}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4,"cache_read":0.25}},"openai/gpt-4-classic-0314":{"id":"openai/gpt-4-classic-0314","name":"GPT-4-Classic-0314","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-26","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":18,"output":72}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":4.5455,"output":27.2727,"cache_read":0.4545}},"xai/grok-3-mini":{"id":"xai/grok-3-mini","name":"Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"xai/grok-4.20-multi-agent":{"id":"xai/grok-4.20-multi-agent","name":"Grok-4.20-Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-code-fast-1":{"id":"xai/grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-22","last_updated":"2025-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok-4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-3":{"id":"xai/grok-3","name":"Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-fast-reasoning":{"id":"xai/grok-4-fast-reasoning","name":"Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok-4.1-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok-4.1-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"xai/grok-4-fast-non-reasoning":{"id":"xai/grok-4-fast-non-reasoning","name":"Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"cerebras/qwen3-32b-cs":{"id":"cerebras/qwen3-32b-cs","name":"qwen3-32b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-15","last_updated":"2025-05-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/llama-3.1-8b-cs":{"id":"cerebras/llama-3.1-8b-cs","name":"Llama-3.1-8B-CS","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.1,"output":0.1}},"cerebras/llama-3.3-70b-cs":{"id":"cerebras/llama-3.3-70b-cs","name":"llama-3.3-70b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/gpt-oss-120b-cs":{"id":"cerebras/gpt-oss-120b-cs","name":"GPT-OSS-120B-CS","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.35,"output":0.75}},"cerebras/qwen3-235b-2507-cs":{"id":"cerebras/qwen3-235b-2507-cs","name":"qwen3-235b-2507-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"runwayml/runway-gen-4-turbo":{"id":"runwayml/runway-gen-4-turbo","name":"Runway-Gen-4-Turbo","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-09","last_updated":"2025-05-09","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}},"runwayml/runway":{"id":"runwayml/runway","name":"Runway","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-11","last_updated":"2024-10-11","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}}}},"modelscope":{"id":"modelscope","env":["MODELSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-inference.modelscope.cn/v1","name":"ModelScope","doc":"https://modelscope.cn/docs/model-service/API-Inference/intro","models":{"ZhipuAI/GLM-4.5":{"id":"ZhipuAI/GLM-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"ZhipuAI/GLM-4.6":{"id":"ZhipuAI/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":98304},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Thinking-2507":{"id":"Qwen/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}}}},"poolside":{"id":"poolside","env":["POOLSIDE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.poolside.ai/v1","name":"Poolside","doc":"https://platform.poolside.ai","models":{"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-m.1":{"id":"poolside/laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"claudinio":{"id":"claudinio","env":["CLAUDINIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.claudin.io/v1","name":"Claudinio","doc":"https://claudin.io","models":{"claudinio":{"id":"claudinio","name":"Claudinio","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-06-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.5,"output":2,"cache_read":0.15}},"claudius":{"id":"claudius","name":"Claudius","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":8,"cache_read":0.9}}}},"novita-ai":{"id":"novita-ai","env":["NOVITA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.novita.ai/openai","name":"NovitaAI","doc":"https://novita.ai/docs/guides/introduction","models":{"paddlepaddle/paddleocr-vl":{"id":"paddlepaddle/paddleocr-vl","name":"PaddleOCR-VL","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.02}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"qwen/qwen3-omni-30b-a3b-instruct":{"id":"qwen/qwen3-omni-30b-a3b-instruct","name":"Qwen3 Omni 30B A3B Instruct","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","video","audio","image"],"output":["text","audio"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-4b-fp8":{"id":"qwen/qwen3-4b-fp8","name":"Qwen3 4B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.03,"output":0.03}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.8,"output":0.8}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30b A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.11,"output":8.45}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"qwen/qwen3-vl-8b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.08,"output":0.5}},"qwen/qwen3-8b-fp8":{"id":"qwen/qwen3-8b-fp8","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.035,"output":0.138}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"qwen/qwen3-vl-30b-a3b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"qwen/qwen3-vl-30b-a3b-thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":1}},"qwen/qwen2.5-7b-instruct":{"id":"qwen/qwen2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.07,"output":0.07}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.38,"output":0.4}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"qwen/qwen3-omni-30b-a3b-thinking":{"id":"qwen/qwen3-omni-30b-a3b-thinking","name":"Qwen3 Omni 30B A3B Thinking","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","audio","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen-mt-plus":{"id":"qwen/qwen-mt-plus","name":"Qwen MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-03","last_updated":"2025-09-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.25,"output":0.75}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"baidu/ernie-4.5-vl-28b-a3b":{"id":"baidu/ernie-4.5-vl-28b-a3b","name":"ERNIE 4.5 VL 28B A3B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2026-06-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":8000},"cost":{"input":0.14,"output":0.56}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"baidu/ernie-4.5-21B-a3b-thinking":{"id":"baidu/ernie-4.5-21B-a3b-thinking","name":"ERNIE-4.5-21B-A3B-Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-21B-a3b":{"id":"baidu/ernie-4.5-21B-a3b","name":"ERNIE 4.5 21B A3B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":8000},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-vl-28b-a3b-thinking":{"id":"baidu/ernie-4.5-vl-28b-a3b-thinking","name":"ERNIE-4.5-VL-28B-A3B-Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.39,"output":0.39}},"kwaipilot/kat-coder-pro":{"id":"kwaipilot/kat-coder-pro","name":"Kat Coder Pro","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-05","last_updated":"2026-01-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-30","last_updated":"2024-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60288,"output":16000},"cost":{"input":0.04,"output":0.17}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.4}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":98304,"output":16384},"cost":{"input":0.119,"output":0.2}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.05,"output":0.1}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.5-air":{"id":"zai-org/glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"zai-org/glm-4.6":{"id":"zai-org/glm-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.6v":{"id":"zai-org/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/autoglm-phone-9b-multilingual":{"id":"zai-org/autoglm-phone-9b-multilingual","name":"AutoGLM-Phone-9B-Multilingual","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.035,"output":0.138}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai-org/glm-5":{"id":"zai-org/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"Mythomax L2 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3200},"cost":{"input":0.09,"output":0.09}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"Wizardlm 2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-24","last_updated":"2024-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"deepseek/deepseek-ocr":{"id":"deepseek/deepseek-ocr","name":"DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-prover-v2-671b":{"id":"deepseek/deepseek-prover-v2-671b","name":"Deepseek Prover V2 671B","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":160000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"Deepseek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-qwen-32b":{"id":"deepseek/deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32000},"cost":{"input":0.3,"output":0.3}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill LLama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-r1-turbo":{"id":"deepseek/deepseek-r1-turbo","name":"DeepSeek R1 (Turbo)\t","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-r1-0528-qwen3-8b":{"id":"deepseek/deepseek-r1-0528-qwen3-8b","name":"DeepSeek R1 0528 Qwen3 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.06,"output":0.09}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"Deepseek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"Deepseek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v3-turbo":{"id":"deepseek/deepseek-v3-turbo","name":"DeepSeek V3 (Turbo)\t","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.4,"output":1.3}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-ocr-2":{"id":"deepseek/deepseek-ocr-2","name":"deepseek/deepseek-ocr-2","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-r1-distill-qwen-14b":{"id":"deepseek/deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.15,"output":0.15}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-08","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-1t":{"id":"inclusionai/ling-2.6-1t","name":"Ling-2.6-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-23","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-flash":{"id":"inclusionai/ling-2.6-flash","name":"Ling-2.6-flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3,"cache_read":0.3}},"xiaomimimo/mimo-v2-pro":{"id":"xiaomimimo/mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.4,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomimimo/mimo-v2.5-pro":{"id":"xiaomimimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":0.522,"output":1.044,"cache_read":0.0043,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.522,"output":1.044,"cache_read":0.0043}}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.05}},"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"meta-llama/llama-3-8b-instruct":{"id":"meta-llama/llama-3-8b-instruct","name":"Llama 3 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.04,"output":0.04}},"meta-llama/llama-4-scout-17b-16e-instruct":{"id":"meta-llama/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-07","last_updated":"2024-12-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"meta-llama/llama-3-70b-instruct":{"id":"meta-llama/llama-3-70b-instruct","name":"Llama3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"nousresearch/hermes-2-pro-llama-3-8b":{"id":"nousresearch/hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-06-27","last_updated":"2024-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"OpenAI: GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.15}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.25}},"sao10K/l3-70b-euryale-v2.1":{"id":"sao10K/l3-70b-euryale-v2.1","name":"L3 70B Euryale V2.1\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-18","last_updated":"2024-06-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}},"sao10K/l3-8b-lunaris":{"id":"sao10K/l3-8b-lunaris","name":"Sao10k L3 8B Lunaris\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.05,"output":0.05}},"sao10K/L3-8B-stheno-v3.2":{"id":"sao10K/L3-8B-stheno-v3.2","name":"L3 8B Stheno V3.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-29","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":32000},"cost":{"input":0.05,"output":0.05}},"sao10K/l31-70b-euryale-v2.2":{"id":"sao10K/l31-70b-euryale-v2.2","name":"L31 70B Euryale V2.2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-07","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"baichuan/baichuan-m2-32b":{"id":"baichuan/baichuan-m2-32b","name":"baichuan-m2-32b","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"baichuan","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.07,"output":0.07}}}},"nebius":{"id":"nebius","env":["NEBIUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenfactory.nebius.com/v1","name":"Nebius Token Factory","doc":"https://docs.tokenfactory.nebius.com/","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":1048000},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":979000,"output":979000},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.15}},"nvidia/Nemotron-3_5-Lightning":{"id":"nvidia/Nemotron-3_5-Lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":3,"cache_read":1}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron-3-Super-120B-A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.3,"output":0.9}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma-3-27b-it","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-10","release_date":"2026-01-20","last_updated":"2026-02-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":110000,"input":100000,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.15,"output":0.5,"cache_read":0.15}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-28","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2026-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":250000,"output":8192},"cost":{"input":0.6,"output":3.6,"cache_read":0.06,"cache_write":0.75}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-10","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"input":40960,"output":0},"cost":{"input":0.01,"output":0}},"NousResearch/Hermes-4-405B":{"id":"NousResearch/Hermes-4-405B","name":"Hermes-4-405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-01-30","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":120000,"output":8192},"cost":{"input":1,"output":3,"reasoning":3,"cache_read":0.1,"cache_write":1.25}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.3,"output":1.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":124000,"output":8192},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.015,"cache_write":0.18}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8000},"cost":{"input":0.95,"output":4}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8000},"cost":{"input":3,"output":15,"cache_read":3}}}},"minimax-cn-coding-plan":{"id":"minimax-cn-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.cn/anthropic/v1","name":"MiniMax Token Plan (minimax.cn)","doc":"https://platform.minimaxi.com/docs/token-plan/intro","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"xiaomi-token-plan-ams":{"id":"xiaomi-token-plan-ams","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-ams.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Europe)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}}}},"zeldoc":{"id":"zeldoc","env":["ZELDOC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.zeldoc.ai/v1","name":"Zeldoc","doc":"https://docs.zeldoc.ai","models":{"zdev":{"id":"zdev","name":"ZDev","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"dinference":{"id":"dinference","env":["DINFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.dinference.com/v1","name":"DInference","doc":"https://dinference.com","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.45,"output":1.65}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":3.89}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.22,"output":0.88}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.75,"output":2.4}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.25,"output":3.89}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08","last_updated":"2025-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.0675,"output":0.27}}}},"pioneer":{"id":"pioneer","env":["PIONEER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pioneer.ai/v1","name":"Pioneer","doc":"https://agent.pioneer.ai/llms.txt","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"devstral-2":{"id":"devstral-2","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.4,"cache_write":0.4}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005,"cache_write":0.05}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"mistral-medium":{"id":"mistral-medium","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.4,"output":2,"cache_read":0.4,"cache_write":0.4}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05,"cache_write":0.1}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.325,"output":1.95,"cache_read":0.065,"cache_write":0.40625}},"claude-opus-5-fast":{"id":"claude-opus-5-fast","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"devstral-small-2":{"id":"devstral-small-2","name":"Devstral Small 2","description":"Compact multimodal coding model for repository exploration, file editing, and software agents","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.1,"cache_write":0.1}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.2,"cache_write":0.4}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"cache_write":1.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03,"cache_write":0.25}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":131072},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25}},"mistral-large-3":{"id":"mistral-large-3","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25,"cache_write":2.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":1,"cache_write":2}},"claude-3-7-sonnet-latest":{"id":"claude-3-7-sonnet-latest","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_read":0.0375,"cache_write":0.234375}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.04,"output":6.24,"cache_read":0.208,"cache_write":1.3}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"ministral-14b":{"id":"ministral-14b","name":"Ministral 14B","description":"Compact multimodal Mistral model for local assistants, edge agents, and efficient tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025,"cache_write":0.25}},"magistral-medium":{"id":"magistral-medium","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":2,"output":5,"cache_read":2,"cache_write":2}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.5,"output":7.5,"cache_read":1.5,"cache_write":1.5}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.083333}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"HuggingFaceTB/SmolLM3-3B-Base":{"id":"HuggingFaceTB/SmolLM3-3B-Base","name":"SmolLM3 3B Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.27,"output":1.12,"cache_read":0.135,"cache_write":0.27}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.0197,"cache_write":0.1}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072},"cost":{"input":0.56,"output":1.68,"cache_read":0.56,"cache_write":0.56}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625,"cache_write":0.435}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01,"cache_write":0.1}},"pioneer/auto":{"id":"pioneer/auto","name":"Pioneer Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2025-06-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":4096}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}},"mistralai/Pixtral-12B-2409":{"id":"mistralai/Pixtral-12B-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03,"cache_read":0.02,"cache_write":0.02}},"mistralai/Codestral-22B-v0.1":{"id":"mistralai/Codestral-22B-v0.1","name":"Codestral-22B-v0.1","description":"Open Mistral code model for fill-in-the-middle and 80+ programming languages","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-29","last_updated":"2024-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.3,"output":0.9,"cache_read":0.3,"cache_write":0.3}},"mistralai/Mistral-7B-Instruct-v0.3":{"id":"mistralai/Mistral-7B-Instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2023-04-30","last_updated":"2023-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"mistralai/Ministral-8B-Instruct-2410":{"id":"mistralai/Ministral-8B-Instruct-2410","name":"Ministral 8B Instruct","description":"Efficient open Mistral edge model for on-device chat and function calling","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small","description":"Open Mistral reasoning model for transparent step-by-step problem solving","family":"magistral","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.05,"cache_write":0.05}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":2.5,"cache_read":0.15,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.09,"output":0.45,"cache_read":0.09,"cache_write":0.09}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-E2B-it":{"id":"google/gemma-4-E2B-it","name":"Gemma 4 E2B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"google/diffusiongemma-26B-A4B-it":{"id":"google/diffusiongemma-26B-A4B-it","name":"DiffusionGemma 26B-A4B IT","description":"Gemini model for general assistance, reasoning, and multimodal workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-12B-it":{"id":"google/gemma-4-12B-it","name":"Gemma 4 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.25,"cache_read":0.25,"cache_write":0.25}},"google/gemma-3-4b-pt":{"id":"google/gemma-3-4b-pt","name":"Gemma 3 4B (Pretrained)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-02-28","last_updated":"2025-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182,"cache_write":0.98}},"zai-org/GLM-5.2-Fast":{"id":"zai-org/GLM-5.2-Fast","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.1,"output":6.6,"cache_read":0.21,"cache_write":2.1}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1040000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":1.4}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2 24B A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-01-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12,"cache_read":0.03,"cache_write":0.03}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":1.2,"cache_read":0.1,"cache_write":0.5}},"fastino/gliguard-LLMGuardrails-300M":{"id":"fastino/gliguard-LLMGuardrails-300M","name":"GLiGuard LLM Guardrails 300M","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-base-v1":{"id":"fastino/gliner2-base-v1","name":"GLiNER2 Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-multi-v1":{"id":"fastino/gliner2-multi-v1","name":"GLiNER2 Multi","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-multi-large-v1":{"id":"fastino/gliner2-multi-large-v1","name":"GLiNER2 Multi Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-privacy-filter-PII-multi":{"id":"fastino/gliner2-privacy-filter-PII-multi","name":"GLiNER2 Privacy Filter PII (Multi)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-large-v1":{"id":"fastino/gliner2-large-v1","name":"GLiNER2 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15,"cache_write":1.25}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-03-31","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3-4B-Instruct-2507":{"id":"Qwen/Qwen3-4B-Instruct-2507","name":"Qwen3 4B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.3,"cache_write":0.3}},"Qwen/Qwen3-1.7B-Base":{"id":"Qwen/Qwen3-1.7B-Base","name":"Qwen3 1.7B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.2,"output":1.2,"cache_read":1.2,"cache_write":1.2}},"Qwen/Qwen3-4B-Base":{"id":"Qwen/Qwen3-4B-Base","name":"Qwen3 4B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.6,"output":0.6,"cache_read":0.6,"cache_write":0.6}},"Qwen/Qwen2.5-Coder-0.5B":{"id":"Qwen/Qwen2.5-Coder-0.5B","name":"Qwen2.5-Coder-0.5B","description":"Tiny open Qwen code model for lightweight completion and on-device coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":1,"cache_read":0.028,"cache_write":0.175}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.3}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.279,"output":1.2,"cache_read":0.279,"cache_write":0.279}},"meta-llama/Llama-3.2-3B":{"id":"meta-llama/Llama-3.2-3B","name":"Llama-3.2-3B","description":"Small open Llama base model for lightweight text generation and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-1B":{"id":"meta-llama/Llama-3.2-1B","name":"Llama-3.2-1B","description":"Compact open Llama base model for lightweight and on-device use","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-06-30","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"meta-llama/Llama-3.2-3B-Instruct":{"id":"meta-llama/Llama-3.2-3B-Instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":80000},"cost":{"input":0.1,"output":0.335,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-1B-Instruct":{"id":"meta-llama/Llama-3.2-1B-Instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":60000},"cost":{"input":0.1,"output":0.201,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.035,"cache_write":0.07}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19,"cache_write":0.95}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.34,"cache_write":0.95}},"moonshotai/Kimi-K3-Fast":{"id":"moonshotai/Kimi-K3-Fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45,"cache_write":4.5}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0.435}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0.14}}}},"helicone":{"id":"helicone","env":["HELICONE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai-gateway.helicone.ai/v1","name":"Helicone","doc":"https://helicone.ai/models","models":{"llama-3.1-8b-instruct-turbo":{"id":"llama-3.1-8b-instruct-turbo","name":"Meta Llama 3.1 8B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03}},"grok-3-mini":{"id":"grok-3-mini","name":"xAI Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"gpt-5-nano":{"id":"gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.049999999999999996,"output":0.39999999999999997,"cache_read":0.005}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"xAI Grok 4.1 Fast Non-Reasoning","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Meta Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.049999999999999996}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"OpenAI GPT-4.1 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998}},"gpt-5-codex":{"id":"gpt-5-codex","name":"OpenAI: GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"claude-3-haiku-20240307":{"id":"claude-3-haiku-20240307","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-03-07","last_updated":"2024-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"glm-4.6":{"id":"glm-4.6","name":"Zai GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.44999999999999996,"output":1.5}},"gpt-5-pro":{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":15,"output":120}},"llama-prompt-guard-2-86m":{"id":"llama-prompt-guard-2-86m","name":"Meta Llama Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":16384},"cost":{"input":0.14,"output":1.4}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1 Codex Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"llama-prompt-guard-2-22m":{"id":"llama-prompt-guard-2-22m","name":"Meta Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1 Codex","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"xAI Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-25","last_updated":"2024-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000},"cost":{"input":0.19999999999999998,"output":1.5,"cache_read":0.02}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi K2 (09/05)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.5,"output":2,"cache_read":0.39999999999999997}},"gemma2-9b-it":{"id":"gemma2-9b-it","name":"Google Gemma 2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-25","last_updated":"2024-06-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.01,"output":0.03}},"chatgpt-4o-latest":{"id":"chatgpt-4o-latest","name":"OpenAI ChatGPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":5,"output":20,"cache_read":2.5}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Google Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998,"cache_write":0.09999999999999999}},"ernie-4.5-21b-a3b-thinking":{"id":"ernie-4.5-21b-a3b-thinking","name":"Baidu Ernie 4.5 21B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-16","last_updated":"2025-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.07,"output":0.28}},"grok-4":{"id":"grok-4","name":"xAI Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-09","last_updated":"2024-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"qwen3-235b-a22b-thinking":{"id":"qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":81920},"cost":{"input":0.3,"output":2.9000000000000004}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":262144},"cost":{"input":0.48,"output":2}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":40960},"cost":{"input":0.29,"output":0.59}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Google Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.19999999999999998}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Anthropic: Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-4.1-mini-2025-04-14":{"id":"gpt-4.1-mini-2025-04-14","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"sonar-reasoning":{"id":"sonar-reasoning","name":"Perplexity Sonar Reasoning","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":5}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"OpenAI GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-09","release_date":"2024-09-30","last_updated":"2024-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Anthropic: Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"OpenAI GPT-OSS 20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.049999999999999996,"output":0.19999999999999998}},"claude-3.5-sonnet-v2":{"id":"claude-3.5-sonnet-v2","name":"Anthropic: Claude 3.5 Sonnet v2","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.95}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"xAI Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.09999999999999999,"output":0.3}},"gpt-5.1":{"id":"gpt-5.1","name":"OpenAI GPT-5.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"grok-3":{"id":"grok-3","name":"xAI Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"OpenAI GPT-5.1 Chat","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"o1-mini":{"id":"o1-mini","name":"OpenAI: o1-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Meta Llama 4 Maverick 17B 128E","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"o1":{"id":"o1","name":"OpenAI: o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"xAI: Grok 4 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Anthropic: Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":20,"output":40}},"llama-guard-4":{"id":"llama-guard-4","name":"Meta Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":1024},"cost":{"input":0.21,"output":0.21}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Anthropic: Claude 4.5 Haiku (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"deepseek-tng-r1t2-chimera":{"id":"deepseek-tng-r1t2-chimera","name":"DeepSeek TNG R1T2 Chimera","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-02","last_updated":"2025-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":163840},"cost":{"input":0.3,"output":1.2}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.03,"output":0.13}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"OpenAI GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Anthropic: Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":0.7999999999999999,"output":4,"cache_read":0.08,"cache_write":1}},"hermes-2-pro-llama-3-8b":{"id":"hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-27","last_updated":"2024-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.14,"output":0.14}},"gpt-4.1":{"id":"gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"sonar":{"id":"sonar","name":"Perplexity Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":1}},"kimi-k2-0711":{"id":"kimi-k2-0711","name":"Kimi K2 (07/11)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.5700000000000001,"output":2.3}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Perplexity Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"claude-opus-4":{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":41000,"output":41000},"cost":{"input":0.08,"output":0.29}},"llama-4-scout":{"id":"llama-4-scout","name":"Meta Llama 4 Scout 17B 16E","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.08,"output":0.3}},"deepseek-v3.1-terminus":{"id":"deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.27,"output":1,"cache_read":0.21600000000000003}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Anthropic: Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Anthropic: Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"mistral-small":{"id":"mistral-small","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.2}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral-Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":6}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":1.5}},"sonar-pro":{"id":"sonar-pro","name":"Perplexity Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":3,"output":15}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Anthropic: Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gpt-5-mini":{"id":"gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"OpenAI GPT-OSS 120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Meta Llama 3.1 8B Instant","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.049999999999999996,"output":0.08}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Google Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.3125,"cache_write":1.25}},"qwen2.5-coder-7b-fast":{"id":"qwen2.5-coder-7b-fast","name":"Qwen2.5 Coder 7B fast","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-15","last_updated":"2024-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.03,"output":0.09}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Anthropic: Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"gpt-5":{"id":"gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Google Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.3}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gemma-3-12b-it":{"id":"gemma-3-12b-it","name":"Google Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.049999999999999996,"output":0.09999999999999999}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Meta Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.59,"output":0.7899999999999999}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Meta Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":0.13,"output":0.39}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"xAI Grok 4 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"o4-mini":{"id":"o4-mini","name":"OpenAI o4 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"o3-mini":{"id":"o3-mini","name":"OpenAI o3 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2023-10","release_date":"2023-10-01","last_updated":"2023-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"OpenAI o3 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}}}},"cloudferro-sherlock":{"id":"cloudferro-sherlock","env":["CLOUDFERRO_SHERLOCK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-sherlock.cloudferro.com/openai/v1/","name":"CloudFerro Sherlock","doc":"https://docs.sherlock.cloudferro.com/","models":{"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"input":180000,"output":16000},"cost":{"input":0.3,"output":1.2}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10-09","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":70000,"output":70000},"cost":{"input":2.92,"output":2.92}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":2.92,"output":2.92}},"speakleash/Bielik-11B-v2.6-Instruct":{"id":"speakleash/Bielik-11B-v2.6-Instruct","name":"Bielik 11B v2.6 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}},"speakleash/Bielik-11B-v3.0-Instruct":{"id":"speakleash/Bielik-11B-v3.0-Instruct","name":"Bielik 11B v3.0 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}}}},"stepfun":{"id":"stepfun","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/v1","name":"StepFun (China)","doc":"https://platform.stepfun.com/docs/zh/overview/concept","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}}}},"unorouter":{"id":"unorouter","env":["UNOROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.unorouter.com/v1","name":"UnoRouter","doc":"https://unorouter.com/models","models":{"deepseek-v4-pro:free":{"id":"deepseek-v4-pro:free","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.819,"output":3.276}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.2675,"output":5.3368}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6001,"output":5.0288}},"qwen3.5-397b-a17b:free":{"id":"qwen3.5-397b-a17b:free","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0625,"output":0.125}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1.8,"output":10.8}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1857,"output":1.1142}},"glm-4.5-flash:free":{"id":"glm-4.5-flash:free","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.2,"output":6}},"step-3.7-flash:free":{"id":"step-3.7-flash:free","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.2:free":{"id":"glm-5.2:free","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"minimax-m2.7:free":{"id":"minimax-m2.7:free","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"gpt-5.4:free":{"id":"gpt-5.4:free","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5.5:free":{"id":"gpt-5.5:free","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.425,"output":2.125}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.8999,"output":1.7999}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.05,"output":8.4}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.44,"output":7.2}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1875,"output":1.125}}}},"coralbricks":{"id":"coralbricks","env":["CORAL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.coralbricks.ai/v1","name":"CoralBricks","doc":"https://www.coralbricks.ai/docs","models":{"glm-5.3-flash-fp4":{"id":"glm-5.3-flash-fp4","name":"GLM 5.3 Flash FP4","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0}},"glm-5.3-fp4":{"id":"glm-5.3-fp4","name":"GLM 5.3 FP4","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.12,"output":4.4,"cache_read":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.12,"output":0.6,"cache_read":0}}}},"hyper":{"id":"hyper","env":["HYPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://hyper.charm.land/v1","name":"Charm Hyper","doc":"https://hyper.charm.land","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":25600},"cost":{"input":0.098,"output":0.334,"cache_read":0.049}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":1.437216,"output":4.311648,"cache_read":0.047907}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.044}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":6553},"cost":{"input":0.484,"output":1.852,"cache_read":0.242}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.152432}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.32664,"output":1.30656,"cache_read":0.064239}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-07-03","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16000},"cost":{"input":1.03436,"output":4.3552,"cache_read":0.206872}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":26214},"cost":{"input":0.6,"output":2.5,"cache_read":0.3}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":64000},"cost":{"input":0.2,"output":0.8,"cache_read":0.04}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16000},"cost":{"input":3.2664,"output":16.332,"cache_read":0.32664}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16332,"output":0.5444,"cache_read":0.031575}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-15","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1.0888,"output":4.40964,"cache_read":0.185096}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-15","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.2,"output":4.8,"cache_read":0.24}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-13","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":13107},"cost":{"input":0.178,"output":0.68,"cache_read":0.089}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.283088}}}},"requesty":{"id":"requesty","env":["REQUESTY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://router.requesty.ai/v1","name":"Requesty","doc":"https://requesty.ai/solution/llm-routing/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7@eu":{"id":"claude-opus-4-7@eu","name":"Claude Opus 4.7 (EU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.34,"cache_read":0.07}},"glm-5.3@eu":{"id":"glm-5.3@eu","name":"GLM-5.3 (EU)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"kimi-k2.7-code@eu":{"id":"kimi-k2.7-code@eu","name":"Kimi K2.7 Code (EU)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.31}},"qwen3.8-2.4T-A95B":{"id":"qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"nemotron-3.5-content-safety":{"id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"mistral-medium-3-5":{"id":"mistral-medium-3-5","name":"mistral-medium-3-5","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"ring-2.6-1t":{"id":"ring-2.6-1t","name":"ring-2.6-1t","description":"Inclusion AI ring-2.6-1t","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"gpt-4.1-mini@eu":{"id":"gpt-4.1-mini@eu","name":"GPT-4.1 mini (EU)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.44,"output":1.76,"cache_read":0.11}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"gemini-3.7-flash@eu":{"id":"gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"leanstral-1-5@eu":{"id":"leanstral-1-5@eu","name":"leanstral-1-5@eu","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813@eu":{"id":"deepseek-v4-pro-0813@eu","name":"DeepSeek V4 Pro 0813 (EU)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"ling-3.0-tiny":{"id":"ling-3.0-tiny","name":"ling-3.0-tiny","description":"Ling-3.0-tiny is an efficient 7.9B parameter MoE model from inclusionAI with only 1.3B active parameters per token. Built for responsive agents, reliable instruction following and multi turn conversation, with a 256K context window, native function calling, prompt caching and switchable Thinking and Instant modes.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"nvidia-nemotron-3-ultra":{"id":"nvidia-nemotron-3-ultra","name":"nvidia-nemotron-3-ultra","description":"NVIDIA Nemotron 3 Ultra is NVIDIA's strongest open-weights reasoning model, positioned near GPT-5.4 Mini (xhigh) and ahead of DeepSeek V4-Flash and Qwen3.5-397B-A17B.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":2.5}},"nvidia-nemotron-3-super-120b-a12b":{"id":"nvidia-nemotron-3-super-120b-a12b","name":"nvidia-nemotron-3-super-120b-a12b","description":"NVIDIA Nemotron 3 Super is a hybrid Mixture-of-Experts (MoE) model engineered for highest compute efficiency and accuracy in multi-agent applications and specialized agentic systems. It is optimized to run many collaborating agents per application on a single GPU, delivering high accuracy for reasoning, tool use, and instruction following.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.5}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":1.2}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-fable-5.1@eu":{"id":"claude-fable-5.1@eu","name":"Claude Fable 5.1 (EU)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.6}},"claude-opus-4-6@eu":{"id":"claude-opus-4-6@eu","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":1.2}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":9,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":9}}},"kat-coder-pro":{"id":"kat-coder-pro","name":"kat-coder-pro","description":"KAT-Coder-Pro V2 by KwaiKAT is a non-reasoning model optimized for agentic coding. It delivers strong performance on reasoning-style tasks while requiring significantly fewer output tokens than peer models. With the 1210 release, it achieved a score of 64 on the Artificial Analysis Intelligence Index, placing it in the global Top 10 and ranking first among all non-reasoning models.","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":1.2}},"gpt-5.6-terra@eu":{"id":"gpt-5.6-terra@eu","name":"GPT-5.6 Terra (EU)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"deepseek-v4.1-flash@eu":{"id":"deepseek-v4.1-flash@eu","name":"DeepSeek V4.1 Flash (EU)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"gpt-5.4@eu":{"id":"gpt-5.4@eu","name":"GPT-5.4 (EU)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"gemini-3.8-flash@eu":{"id":"gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-nano@eu":{"id":"gpt-5-nano@eu","name":"GPT-5 Nano (EU)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.055,"output":0.44,"cache_read":0.0055}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"claude-sonnet-5@eu":{"id":"claude-sonnet-5@eu","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":7,"cache_read":0.15}},"nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"gpt-5.5@eu":{"id":"gpt-5.5@eu","name":"GPT-5.5 (EU)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"seed-1.8":{"id":"seed-1.8","name":"seed-1.8","description":"Optimized specifically for multimodal agent scenarios. It features enhanced agent capabilities, upgraded multimodal comprehension, and more flexible context management.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.1}},"gpt-5-mini@eu":{"id":"gpt-5-mini@eu","name":"GPT-5 Mini (EU)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.275,"output":2.2,"cache_read":0.0275}},"gpt-4.1-nano@eu":{"id":"gpt-4.1-nano@eu","name":"GPT-4.1 nano (EU)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.11,"output":0.44,"cache_read":0.0275}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"claude-fable-5@eu":{"id":"claude-fable-5@eu","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"ling-2.6-1t":{"id":"ling-2.6-1t","name":"ling-2.6-1t","description":"Inclusion AI ling-2.6-1t","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"seed-2.0-pro":{"id":"seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"glm-5.1@eu":{"id":"glm-5.1@eu","name":"GLM-5.1 (EU)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"qwen3.8-flash-next@eu":{"id":"qwen3.8-flash-next@eu","name":"Qwen3.8 Flash Next (EU)","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"nemotron-3-ultra-nvfp4":{"id":"nemotron-3-ultra-nvfp4","name":"nemotron-3-ultra-nvfp4","description":"Nemotron-3-Ultra-550B-A55B-NVFP4 is a frontier-scale large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities. It is optimized for the most demanding workloads, including complex multi-step agents, long-context analysis, and high-accuracy reasoning over code, math, and science. The model employs a hybrid Latent Mixture-of-Experts (LatentMoE) architecture, utilizing interleaved Mamba-2 and MoE layers, along with select Attention layers. Like the Super model, the Ultra model incorporates Multi-Token Prediction (MTP) layers for faster text generation and improved quality, and it is trained using an NVFP4 pre-training recipe to maximize compute efficiency. The model has 55B active parameters and 550B parameters in total.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax-m3@eu":{"id":"minimax-m3@eu","name":"MiniMax-M3 (EU)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"devstral-latest@eu":{"id":"devstral-latest@eu","name":"devstral-latest@eu","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.583}},"mistral-medium-3-5@eu":{"id":"mistral-medium-3-5@eu","name":"mistral-medium-3-5@eu","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"nemotron-lightning-3.5-30b-a3b":{"id":"nemotron-lightning-3.5-30b-a3b","name":"nemotron-lightning-3.5-30b-a3b","description":"Nemotron-Lightning-3.5-30B-A3B is a 30B-parameter Mixture-of-Experts language model (3B active) from NVIDIA's Nemotron-H family, built on a hybrid Mamba-Transformer architecture for efficient long-context inference. Like other models in the family, it responds to queries by first generating a reasoning trace and then concluding with a final response, with reasoning behavior configurable through a flag in the chat template. It includes a multi-token prediction (MTP) speculative decoding head for low-latency serving.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-15","last_updated":"2026-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"seed-2.0-code":{"id":"seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"mistral-medium-latest@eu":{"id":"mistral-medium-latest@eu","name":"Mistral Medium (latest) (EU)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"gpt-5.6-sol@eu":{"id":"gpt-5.6-sol@eu","name":"GPT-5.6 Sol (EU)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"glm-5.2-fast","description":"GLM-5.2 introduces a robust 1M-token context and advanced, multi-effort coding capabilities to significantly enhance performance on long-horizon tasks. Its new IndexShare architecture and improved MTP layer simultaneously boost efficiency by reducing per-token FLOPs and increasing speculative decoding lengths. A 743B-parameter model in Zhipu AI's GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-13","last_updated":"2026-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"claude-sonnet-4-6@eu":{"id":"claude-sonnet-4-6@eu","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"laguna-m.1":{"id":"laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"leanstral-1-5":{"id":"leanstral-1-5","name":"leanstral-1-5","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.3-flash@eu":{"id":"glm-5.3-flash@eu","name":"GLM-5.3-Flash (EU)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"laguna-xs.2":{"id":"laguna-xs.2","name":"Laguna XS.2","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"nemotron-3-nano-omni","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gpt-5@eu":{"id":"gpt-5@eu","name":"GPT-5 (EU)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"seed-2.0-mini":{"id":"seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"kimi-k2.6@eu":{"id":"kimi-k2.6@eu","name":"Kimi K2.6 (EU)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"nemotron-3-nano-omni@eu":{"id":"nemotron-3-nano-omni@eu","name":"nemotron-3-nano-omni@eu","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"ling-2.6-flash":{"id":"ling-2.6-flash","name":"ling-2.6-flash","description":"Inclusion AI ling-2.6-flash","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3}},"deepseek-v4-pro@eu":{"id":"deepseek-v4-pro@eu","name":"DeepSeek V4 Pro (EU)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"claude-sonnet-4@eu":{"id":"claude-sonnet-4@eu","name":"Claude Sonnet 4 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"gemini-3.5-flash-lite@eu":{"id":"gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.033}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"claude-opus-5@eu":{"id":"claude-opus-5@eu","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"glm-5.2@eu":{"id":"glm-5.2@eu","name":"GLM-5.2 (EU)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"claude-haiku-4-5@eu":{"id":"claude-haiku-4-5@eu","name":"Claude Haiku 4.5 (latest) (EU)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"gpt-4o-mini@eu":{"id":"gpt-4o-mini@eu","name":"GPT-4o mini (EU)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.165,"output":0.66,"cache_read":0.0825}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"devstral-latest":{"id":"devstral-latest","name":"devstral-latest","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"kimi-k3@eu":{"id":"kimi-k3@eu","name":"Kimi K3 (EU)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.032,"cache_write":0.4}},"gemini-2.5-flash-lite@eu":{"id":"gemini-2.5-flash-lite@eu","name":"Gemini 2.5 Flash-Lite (EU)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.18333}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4.2-beta":{"id":"grok-4.2-beta","name":"grok-4.2-beta","description":"Grok 4.20 Beta is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently precise and truthful responses.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":2,"output":6,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.4,"cache_write":4}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"claude-sonnet-4-5@eu":{"id":"claude-sonnet-4-5@eu","name":"Claude Sonnet 4.5 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125,"tiers":[{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25}}},"mistral-small-2603@eu":{"id":"mistral-small-2603@eu","name":"Mistral Small 4 (EU)","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"qwen3.5-2b","description":"Qwen3.5-2B is a compact yet capable model from Alibaba's Qwen3.5 series. It features a 262K token context window, support for 201 languages, thinking/reasoning mode, and tool calling for agentic workflows. A strong choice for prototyping, fine-tuning, and efficient multilingual deployments.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.02,"output":0.1}},"claude-opus-4-5@eu":{"id":"claude-opus-4-5@eu","name":"Claude Opus 4.5 (latest) (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":30}},"gemini-3.1-flash-lite@eu":{"id":"gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.275,"output":1.65,"cache_read":0.0275,"cache_write":0.091663}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gemini-2.5-flash@eu":{"id":"gemini-2.5-flash@eu","name":"Gemini 2.5 Flash (EU)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.55}},"gemini-2.5-pro@eu":{"id":"gemini-2.5-pro@eu","name":"Gemini 2.5 Pro (EU)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":2.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"nemotron-3.5-lightning-30b-a3b":{"id":"nemotron-3.5-lightning-30b-a3b","name":"nemotron-3.5-lightning-30b-a3b","description":"NVIDIA Nemotron 3.5 Lightning 30B-A3B is a hybrid Mamba-2 + MoE + Attention model with 30B total and 3B active parameters, pre-trained on over 20T tokens with an NVFP4 recipe and Multi-Token Prediction for fast generation. Up to 1M token context for long-running autonomous agents, sub-agent workhorse deployments, and agentic workflows. Supports reasoning and tool calling. English and coding languages plus Spanish, French, German, Italian, and Japanese. Open weights under the OpenMDW License Agreement v1.1. Part of the NVIDIA Nemotron family.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"inkling-256k":{"id":"inkling-256k","name":"inkling-256k","description":"Inkling 256K is the extended context variant of Inkling, a large MoE hybrid reasoning model from Thinking Machines with audio and vision input support and a 256K context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"gemini-3.5-flash@eu":{"id":"gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.65,"output":9.9,"cache_read":0.165,"cache_write":1.7413}},"gpt-5.6-luna@eu":{"id":"gpt-5.6-luna@eu","name":"GPT-5.6 Luna (EU)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022}},"claude-opus-4-8@eu":{"id":"claude-opus-4-8@eu","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"o4-mini@eu":{"id":"o4-mini@eu","name":"o4-mini (EU)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.21,"output":4.84,"cache_read":0.3025}},"gpt-4.1@eu":{"id":"gpt-4.1@eu","name":"GPT-4.1 (EU)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.2,"output":8.8,"cache_read":0.55}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"deepseek-v4-flash-0731@eu":{"id":"deepseek-v4-flash-0731@eu","name":"DeepSeek V4 Flash 0731 (EU)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"qwen3.8-2.4T-A95B@eu":{"id":"qwen3.8-2.4T-A95B@eu","name":"Qwen3.8 2.4T A95B (EU)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"gpt-5.1@eu":{"id":"gpt-5.1@eu","name":"GPT-5.1 (EU)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":33,"cache_read":0.55}}}},"llmtr":{"id":"llmtr","env":["LLMTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llmtr.com/v1","name":"LLMTR","doc":"https://llmtr.com/docs","models":{"medgemma-4b":{"id":"medgemma-4b","name":"MedGemma 4B","description":"Multimodal medical-domain Gemma variant for text and image analysis","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":3,"output":5}},"muse-glimmer-30b-tr":{"id":"muse-glimmer-30b-tr","name":"Muse Glimmer 30B (TR)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"gemma-4":{"id":"gemma-4","name":"Gemma 4","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"magibu-11b-v8":{"id":"magibu-11b-v8","name":"Magibu 11B v8","description":"Turkish-language chat model for instruction following and assistant flows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.1,"output":0.5}},"qwen3-6-35b":{"id":"qwen3-6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":5,"output":10}},"trendyol-asure-12b":{"id":"trendyol-asure-12b","name":"Trendyol Asure 12B","description":"Turkish-language multimodal instruct model built on Gemma 3 12B for e-commerce text, chat, and image-text tasks","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-19","last_updated":"2026-02-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.1,"output":0.5,"cache_read":0.025}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"cost":{"input":0.2,"output":1.6}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mimo/mimo-v2.5":{"id":"mimo/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28}},"mimo/mimo-v2.5-pro":{"id":"mimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.1}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.58,"output":1.44}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.87,"output":4.68}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2}},"publicai/apertus-8b-instruct":{"id":"publicai/apertus-8b-instruct","name":"Apertus 8B Instruct","description":"Fully open 8B multilingual LLM supporting 1800+ languages with 65K context. Trained on compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.1,"output":0.2}},"publicai/apertus-70b-instruct":{"id":"publicai/apertus-70b-instruct","name":"Apertus 70B Instruct","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.82,"output":2.92}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.03,"output":0.12}},"upstage/solar-pro3":{"id":"upstage/solar-pro3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro2":{"id":"upstage/solar-pro2","name":"Solar Pro 2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.15,"output":0.6}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"xiaomi":{"id":"xiaomi","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.xiaomimimo.com/v1","name":"Xiaomi","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro-ultraspeed":{"id":"mimo-v2.5-pro-ultraspeed","name":"MiMo-V2.5-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-06-08","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":1.305,"output":2.61,"cache_read":0.0108}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"MiMo-V2-Flash","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo-V2-Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}}}},"huggingface":{"id":"huggingface","env":["HF_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://router.huggingface.co/v1","name":"Hugging Face","doc":"https://huggingface.co/docs/inference-providers","models":{"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":8192},"cost":{"input":0.4,"output":1.3}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":3,"output":5}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32768},"cost":{"input":0.7,"output":2.5}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.28,"output":0.4}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"zai-org/GLM-4.6V-Flash":{"id":"zai-org/GLM-4.6V-Flash","name":"GLM-4.6V-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-4.5V":{"id":"zai-org/GLM-4.5V","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.5,"output":1.2}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.07,"output":0.26}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3-Coder-Next":{"id":"Qwen/Qwen3-Coder-Next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3-235B-A22B":{"id":"Qwen/Qwen3-235B-A22B","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":6.25}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.855,"output":2.565}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":2,"output":2}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3.6}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.29,"output":0.59}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":3}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.47,"output":3.19}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.95}},"Qwen/Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen 3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.01,"output":0}},"Qwen/Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen/Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next-80B-A3B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3-Embedding-4B":{"id":"Qwen/Qwen3-Embedding-4B","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"Qwen/Qwen2.5-Coder-32B-Instruct":{"id":"Qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen2.5-Coder-32B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.06,"output":0.2}},"MiniMaxAI/MiniMax-M2":{"id":"MiniMaxAI/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-10","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.06,"output":0.06}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.59,"output":0.79}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":0.69}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi-K2-Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi-K2-Instruct-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1,"output":3}},"moonshotai/Kimi-K2-Instruct":{"id":"moonshotai/Kimi-K2-Instruct","name":"Kimi-K2-Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-14","last_updated":"2025-07-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":3}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15}},"tencent/Hy4-preview":{"id":"tencent/Hy4-preview","name":"Hy4 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.834,"output":2.501}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"MiMo model for long-context reasoning, perception, and agentic tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.4,"output":2}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.3}}}},"zhipuai-coding-plan":{"id":"zhipuai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/coding/paas/v4","name":"Zhipu AI Coding Plan","doc":"https://docs.bigmodel.cn/cn/coding-plan/overview","models":{"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}}}},"daoxe":{"id":"daoxe","env":["DAOXE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://daoxe.com/v1","name":"DaoXE","doc":"https://daoxe.com/pricing","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"crossmodel":{"id":"crossmodel","env":["CROSSMODEL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.crossmodel.ai/v1","name":"CrossModel","doc":"https://www.crossmodel.ai/docs","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.88,"output":5.63,"cache_read":0.375,"cache_write":2.35}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.32,"output":1.88,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57}}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.04,"output":0.13,"cache_read":0.01,"cache_write":0.04,"tiers":[{"input":0.1,"output":0.37,"cache_read":0.02,"cache_write":0.12,"tier":{"type":"context","size":32000}},{"input":0.19,"output":0.74,"cache_read":0.04,"cache_write":0.24,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.13,"output":0.43,"cache_read":0.016,"cache_write":0.13}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.19,"output":1.13,"cache_read":0.019,"cache_write":0.24,"tiers":[{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94}}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.13,"output":0.43,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.88,"output":5.63,"cache_read":0.23,"cache_write":2.35}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.25,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2}}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.16,"output":0.32,"cache_read":0.004,"cache_write":0.16}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.47,"output":0.94,"cache_read":0.005,"cache_write":0.47}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.42}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.33,"tiers":[{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66}}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.215,"output":3.645,"cache_read":0.0405,"cache_write":1.215}},"gemini/gemini-3.1-pro-preview":{"id":"gemini/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":4}}},"gemini/gemini-2.5-flash-lite":{"id":"gemini/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.1}},"gemini/gemini-3.6-flash":{"id":"gemini/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3.5-flash":{"id":"gemini/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.5}},"gemini/gemini-3.5-flash-lite":{"id":"gemini/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gemini/gemini-3-flash-preview":{"id":"gemini/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.5}},"gemini/gemini-3.8-flash":{"id":"gemini/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3.7-flash":{"id":"gemini/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-2.5-pro":{"id":"gemini/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}}},"gemini/gemini-2.5-flash":{"id":"gemini/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.6,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6,"cache_write":4}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"cache_write":1,"tiers":[{"input":2,"output":4,"cache_read":0.4,"cache_write":2,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4,"cache_write":2}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5}}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":10}}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.16,"output":0.64,"cache_read":0.04,"cache_write":0.16}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0.96,"output":2.88,"cache_read":0.048,"cache_write":0.96}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.47,"output":2.16,"cache_read":0.1,"cache_write":0.47,"tiers":[{"input":0.62,"output":2.47,"cache_read":0.13,"cache_write":0.62,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0.15}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.16,"cache_write":0.6,"tiers":[{"input":0.8,"output":3.4,"cache_read":0.2,"cache_write":0.8,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1,"output":3.8,"cache_read":0.2,"cache_write":1,"tiers":[{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":3.7,"cache_read":0.18,"cache_write":0.9,"tiers":[{"input":1.1,"output":4.3,"cache_read":0.27,"cache_write":1.1,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}}}},"minimax":{"id":"minimax","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax (minimax.io)","doc":"https://platform.minimax.io/docs/guides/quickstart","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}}}},"salad-cloud":{"id":"salad-cloud","env":["SALAD_CLOUD_API_KEY"],"npm":"@saladtechnologies-oss/ai-sdk-provider","name":"SaladCloud AI Gateway","doc":"https://docs.salad.com/ai-gateway/explanation/overview","models":{"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen MoE for agentic tasks, complex reasoning, code generation, and instruction following","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.09,"output":0.6}}}},"aki-io":{"id":"aki-io","env":["AKI_IO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://aki.io/v1","name":"AKI.IO","doc":"https://aki.io/docs/","models":{"gemma4-26b":{"id":"gemma4-26b","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.5}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.2,"cache_read":0.1}},"glm5.3-754b":{"id":"glm5.3-754b","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":81920},"cost":{"input":1,"output":3.5,"cache_read":0.25}},"deepseek-v4-flash-0731-284b":{"id":"deepseek-v4-flash-0731-284b","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":81920},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"mistral4-119b":{"id":"mistral4-119b","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.2,"output":0.6}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.55}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.15,"output":0.5}}}},"trustedrouter":{"id":"trustedrouter","env":["TRUSTEDROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.trustedrouter.com/v1","name":"TrustedRouter","doc":"https://trustedrouter.com/docs","models":{"trustedrouter/zdr":{"id":"trustedrouter/zdr","name":"Zero Data Retention","description":"TrustedRouter privacy routing alias that prefers zero data retention model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth":{"id":"trustedrouter/synth","name":"Synth","description":"TrustedRouter synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/e2e":{"id":"trustedrouter/e2e","name":"End-to-End Encrypted","description":"TrustedRouter privacy routing alias for end-to-end encrypted provider routes where available.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth-code":{"id":"trustedrouter/synth-code","name":"Synth Code","description":"TrustedRouter code synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/fast":{"id":"trustedrouter/fast","name":"Fast","description":"TrustedRouter speed routing alias that prefers low-latency healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/cheap":{"id":"trustedrouter/cheap","name":"Cheap","description":"TrustedRouter low-cost routing alias that prefers inexpensive healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/auto":{"id":"trustedrouter/auto","name":"Auto","description":"TrustedRouter automatic routing alias that chooses a healthy supported model endpoint for the request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}}}},"alibaba":{"id":"alibaba","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope-intl.aliyuncs.com/compatible-mode/v1","name":"Alibaba","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":1.4,"output":5.6}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":5}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":6}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.1,"output":0.4,"input_audio":6.76}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.16,"output":0.49}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4,"reasoning":4.2}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.27,"output":1.07,"input_audio":4.44,"output_audio":8.89}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28,"cache_write":0}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.05,"output":0.2,"reasoning":0.5}},"qwen3-livetranslate-flash-realtime":{"id":"qwen3-livetranslate-flash-realtime","name":"Qwen3-LiveTranslate Flash Realtime","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":10,"output":10,"input_audio":10,"output_audio":38}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"reasoning":2.4}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2025-04-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.72,"output":0.72}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.175,"output":0.7}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25,"tiers":[{"input":0.75,"output":3.75,"tier":{"type":"context","size":32000}},{"input":1.2,"output":6,"tier":{"type":"context","size":128000}}]}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.035,"output":0.035}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.43,"output":1.66,"input_audio":3.81,"output_audio":15.11}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.8,"output":8.4}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.52,"output":1.99,"input_audio":4.57,"output_audio":18.13}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.05}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"qwen-plus-character-ja":{"id":"qwen-plus-character-ja","name":"Qwen Plus Character (Japanese)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":512},"cost":{"input":0.5,"output":1.4}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.7,"reasoning":2.1}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-04","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.07,"output":0.27,"input_audio":4.44,"output_audio":8.89}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"tiers":[{"input":2.7,"output":13.5,"tier":{"type":"context","size":32000}},{"input":4.5,"output":22.5,"tier":{"type":"context","size":128000}}]}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.7,"output":2.8}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":2.46,"output":7.37}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.2,"output":4.8}}}},"nvidia":{"id":"nvidia","env":["NVIDIA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://integrate.api.nvidia.com/v1","name":"Nvidia","doc":"https://docs.api.nvidia.com/nim/","models":{"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen/qwen-image":{"id":"qwen/qwen-image","name":"Qwen Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":66536},"cost":{"input":0,"output":0}},"qwen/qwen-image-edit":{"id":"qwen/qwen-image-edit","name":"Qwen Image Edit","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen/qwen2.5-coder-32b-instruct":{"id":"qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32b Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-06","last_updated":"2024-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"stepfun-ai/step-3.7-flash":{"id":"stepfun-ai/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-pro-0813":{"id":"deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-large-3-675b-instruct-2512":{"id":"mistralai/mistral-large-3-675b-instruct-2512","name":"Mistral Large 3 675B Instruct 2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"mistralai/mistral-nemotron":{"id":"mistralai/mistral-nemotron","name":"mistral-nemotron","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":13108},"cost":{"input":0,"output":0}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B Instruct 2512","description":"Compact Mistral VLM for chat and instruction-based workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"mistral-small-4-119b-2603","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x7b-instruct":{"id":"mistralai/mixtral-8x7b-instruct","name":"Mistral: Mixtral 8x7B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2023-12-10","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3.5-128b":{"id":"mistralai/mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mistralai/mistral-7b-instruct-v0.3":{"id":"mistralai/mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3-instruct":{"id":"mistralai/mistral-medium-3-instruct","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0,"output":0}},"mistralai/magistral-small-2506":{"id":"mistralai/magistral-small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0,"output":0}},"nvidia/streampetr":{"id":"nvidia/streampetr","name":"streampetr","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1.5":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/usdcode":{"id":"nvidia/usdcode","name":"usdcode","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":-1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer1-7b":{"id":"nvidia/cosmos-transfer1-7b","name":"cosmos-transfer1-7b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-13","last_updated":"2025-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-voicechat":{"id":"nvidia/nemotron-voicechat","name":"nemotron-voicechat","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/studiovoice":{"id":"nvidia/studiovoice","name":"studiovoice","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-03","last_updated":"2025-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-content-safety":{"id":"nvidia/nemotron-3-content-safety","name":"nemotron-3-content-safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer2_5-2b":{"id":"nvidia/cosmos-transfer2_5-2b","name":"cosmos-transfer2.5-2b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/bevformer":{"id":"nvidia/bevformer","name":"bevformer","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-vl-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-vl-8b-v1","name":"Llama 3.1 Nemotron Nano VL 8B v1","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-10","last_updated":"2025-04-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"nvidia/magpie-tts-zeroshot":{"id":"nvidia/magpie-tts-zeroshot","name":"magpie-tts-zeroshot","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-06-12","modalities":{"input":["text","audio"],"output":["audio"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nemotron Nano 12B v2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"nvidia/sparsedrive":{"id":"nvidia/sparsedrive","name":"sparsedrive","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-nemotron-embed-vl-1b-v2":{"id":"nvidia/llama-nemotron-embed-vl-1b-v2","name":"llama-nemotron-embed-vl-1b-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/synthetic-video-detector":{"id":"nvidia/synthetic-video-detector","name":"synthetic-video-detector","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"nvidia/llama-nemotron-rerank-vl-1b-v2":{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2","name":"llama-nemotron-rerank-vl-1b-v2","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/usdvalidate":{"id":"nvidia/usdvalidate","name":"usdvalidate","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-24","last_updated":"2025-01-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/active-speaker-detection":{"id":"nvidia/active-speaker-detection","name":"Active Speaker Detection","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-ultra-253b-v1":{"id":"nvidia/llama-3.1-nemotron-ultra-253b-v1","name":"Llama 3.1 Nemotron Ultra 253B","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"nvidia/llama-3_2-nemoretriever-300m-embed-v1":{"id":"nvidia/llama-3_2-nemoretriever-300m-embed-v1","name":"llama-3_2-nemoretriever-300m-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-24","last_updated":"2025-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/nv-embedcode-7b-v1":{"id":"nvidia/nv-embedcode-7b-v1","name":"nv-embedcode-7b-v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-17","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-safety-guard-8b-v3":{"id":"nvidia/llama-3.1-nemotron-safety-guard-8b-v3","name":"llama-3.1-nemotron-safety-guard-8b-v3","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-mini-4b-instruct":{"id":"nvidia/nemotron-mini-4b-instruct","name":"nemotron-mini-4b-instruct","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-08-21","last_updated":"2024-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-predict1-5b":{"id":"nvidia/cosmos-predict1-5b","name":"cosmos-predict1-5b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-content-safety-reasoning-4b":{"id":"nvidia/nemotron-content-safety-reasoning-4b","name":"nemotron-content-safety-reasoning-4b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/riva-translate-4b-instruct-v1.1":{"id":"nvidia/riva-translate-4b-instruct-v1.1","name":"riva-translate-4b-instruct-v1_1","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nvidia-nemotron-nano-9b-v2":{"id":"nvidia/nvidia-nemotron-nano-9b-v2","name":"nvidia-nemotron-nano-9b-v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/gliner-pii":{"id":"nvidia/gliner-pii","name":"gliner-pii","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-8b-v1","name":"Llama 3.1 Nemotron Nano 8B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"nvidia/nv-embed-v1":{"id":"nvidia/nv-embed-v1","name":"nv-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-07","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/rerank-qa-mistral-4b":{"id":"nvidia/rerank-qa-mistral-4b","name":"rerank-qa-mistral-4b","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-17","last_updated":"2025-01-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.15}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1","name":"Llama 3.3 Nemotron Super 49B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"nemotron-3-nano-30b-a3b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-70b-instruct":{"id":"nvidia/llama-3.1-nemotron-70b-instruct","name":"Llama 3.1 Nemotron 70B Instruct","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-reason2-8b":{"id":"nvidia/cosmos-reason2-8b","name":"Cosmos Reason2 8B","description":"Vision language model for physical-world understanding with structured reasoning on video and images","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3n-e2b-it":{"id":"google/gemma-3n-e2b-it","name":"Gemma 3n E2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-12","last_updated":"2025-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/google-paligemma":{"id":"google/google-paligemma","name":"paligemma","description":"Gemini multimodal model for text, image, audio, video, and document tasks","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-14","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"google/gemma-2-2b-it":{"id":"google/gemma-2-2b-it","name":"Gemma 2 2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma-4-31B-IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3n-e4b-it":{"id":"google/gemma-3n-e4b-it","name":"Gemma 3n E4b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0,"output":0}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-guard-4-12b":{"id":"meta/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"meta/llama-3.2-90b-vision-instruct":{"id":"meta/llama-3.2-90b-vision-instruct","name":"Llama-3.2-90B-Vision-Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0,"output":0}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/esmfold":{"id":"meta/esmfold","name":"esmfold","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-15","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-4-maverick-17b-128e-instruct":{"id":"meta/llama-4-maverick-17b-128e-instruct","name":"Llama 4 Maverick 17b 128e Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-02","release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"meta/esm2-650m":{"id":"meta/esm2-650m","name":"esm2-650m","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-29","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11b Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-26","last_updated":"2024-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"bytedance/seed-oss-36b-instruct":{"id":"bytedance/seed-oss-36b-instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0,"output":0}},"sarvamai/sarvam-m":{"id":"sarvamai/sarvam-m","name":"sarvam-m","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"microsoft/phi-4-multimodal-instruct":{"id":"microsoft/phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0,"output":0}},"microsoft/phi-4-mini-instruct":{"id":"microsoft/phi-4-mini-instruct","name":"Phi-4-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"minimaxai/minimax-m2.7":{"id":"minimaxai/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0,"output":0}},"baai/bge-m3":{"id":"baai/bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0,"output":0}},"abacusai/dracarys-llama-3.1-70b-instruct":{"id":"abacusai/dracarys-llama-3.1-70b-instruct","name":"dracarys-llama-3.1-70b-instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-11","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS-120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-04","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"moonshotai/kimi-k2-instruct-0905":{"id":"moonshotai/kimi-k2-instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"upstage/solar-10.7b-instruct":{"id":"upstage/solar-10.7b-instruct","name":"solar-10.7b-instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-05","last_updated":"2025-04-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-kontext-dev":{"id":"black-forest-labs/flux_1-kontext-dev","name":"FLUX.1-Kontext-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-schnell":{"id":"black-forest-labs/flux_1-schnell","name":"FLUX.1-schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-07","release_date":"2024-08-01","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":77,"input":77,"output":0},"cost":{"input":0,"output":0}},"black-forest-labs/flux_2-klein-4b":{"id":"black-forest-labs/flux_2-klein-4b","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-14","last_updated":"2026-01-31","modalities":{"input":["image","text"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"black-forest-labs/flux.1-dev":{"id":"black-forest-labs/flux.1-dev","name":"FLUX.1-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0},"cost":{"input":0,"output":0}}}},"jiekou":{"id":"jiekou","env":["JIEKOU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jiekou.ai/openai","name":"Jiekou.AI","doc":"https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"grok-4-1-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gpt-5-codex":{"id":"gpt-5-codex","name":"gpt-5-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":13.5,"output":108}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"gpt-5.1-codex-mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"gpt-5.1-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"grok-code-fast-1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.18,"output":1.35}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"gpt-5.2-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"gemini-2.5-pro-preview-06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"gpt-5.2-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":18.9,"output":151.2}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":10.8}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"gpt-5-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"gemini-2.5-flash-lite-preview-06-17","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","video","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"claude-opus-4-20250514","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"gpt-5.1-codex-max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.36}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"claude-opus-4-6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":64000},"cost":{"input":0.9,"output":4.5}},"grok-4-0709":{"id":"grok-4-0709","name":"grok-4-0709","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.7,"output":13.5}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"gemini-2.5-flash-preview-05-20","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":0.135,"output":3.15}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.125,"output":9}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.575,"output":12.6}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"claude-sonnet-4-20250514","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.27,"output":2.25}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":1.1,"output":4.4}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":4.5,"output":22.5}},"o3":{"id":"o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":10,"output":40}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":3}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"qwen/qwen3-coder-next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.15,"output":0.8}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.2}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":131071}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.28,"output":1.14}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32767}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":262143}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}}}},"frogbot":{"id":"frogbot","env":["FROGBOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://app.frogbot.ai/api/v1","name":"FrogBot","doc":"https://docs.frogbot.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.2}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2-5":{"id":"minimax-m2-5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-01-15","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"zai-glm-5-1":{"id":"zai-glm-5-1","name":"Z.AI GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-20","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":8192},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek v4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":1.74,"output":3.48,"cache_read":0.14}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-07-17","last_updated":"2025-07-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}}}},"ovhcloud":{"id":"ovhcloud","env":["OVHCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://oai.endpoints.kepler.ai.cloud.ovh.net/v1","name":"OVHcloud AI Endpoints","doc":"https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//","models":{"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"Qwen3Guard-Gen-0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.18}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.47,"output":3.19}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.18}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen2.5-VL-72B-Instruct","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":1.01,"output":1.01}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder-30B-A3B-Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.26}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-18","last_updated":"2026-05-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":4.25}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.47,"output":3.19}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral-Nemo-Instruct-2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.14,"output":0.14}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral-Small-3.2-24B-Instruct-2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-16","last_updated":"2025-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.31}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.11,"output":0.11}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"Qwen3Guard-Gen-8B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.09,"output":0.47}},"meta-llama-3_3-70b-instruct":{"id":"meta-llama-3_3-70b-instruct","name":"Meta-Llama-3_3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.74,"output":0.74}}}},"xpersona":{"id":"xpersona","env":["XPERSONA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://www.xpersona.co/v1","name":"Xpersona","doc":"https://www.xpersona.co/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":5.55,"reasoning":5.55,"cache_read":0.09}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"xpersona-gpt-5.5":{"id":"xpersona-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18,"reasoning":18,"cache_read":0.3}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.75,"output":6,"reasoning":6,"cache_read":0.075}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.55,"output":12.2,"reasoning":12.2,"cache_read":0.155}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18.5,"reasoning":18.5,"cache_read":0.3}},"xpersona-frieren-coder":{"id":"xpersona-frieren-coder","name":"Xpersona Frieren 1","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-01","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":1.5,"output":6,"reasoning":6,"cache_read":0.15}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.375,"output":4,"reasoning":4,"cache_read":0.0375}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3.7,"reasoning":3.7,"cache_read":0.06}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.5,"output":9.25,"reasoning":9.25,"cache_read":0.15}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":2,"reasoning":2,"cache_read":0.15}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}}}},"anthropic":{"id":"anthropic","env":["ANTHROPIC_API_KEY"],"npm":"@ai-sdk/anthropic","name":"Anthropic","doc":"https://docs.anthropic.com/en/docs/about-claude/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-04","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-14","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-07","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}}}},"google":{"id":"google","env":["GOOGLE_API_KEY","GOOGLE_GENERATIVE_AI_API_KEY","GEMINI_API_KEY"],"npm":"@ai-sdk/google","name":"Google","doc":"https://ai.google.dev/gemini-api/docs/models","models":{"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3.1-flash-lite-image":{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.25,"output":30}},"lyria-3-clip-preview":{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Music generation model for short 30-second clips, loops, and previews from text or image prompts","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30,"cache_read":0.075}},"deep-research-max-preview-04-2026":{"id":"deep-research-max-preview-04-2026","name":"Deep Research Max Preview (Apr-21-2026)","description":"Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":120}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"deep-research-preview-04-2026":{"id":"deep-research-preview-04-2026","name":"Deep Research Preview (Apr-21-2026)","description":"Agentic model for autonomous multi-step research, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"gemini-2.5-computer-use-preview-10-2025":{"id":"gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview 10-2025","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1.25,"output":10,"tiers":[{"input":2.5,"output":15,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gemini-3.1-flash-live-preview":{"id":"gemini-3.1-flash-live-preview","name":"Gemini 3.1 Flash Live Preview","description":"High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image","video","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.75,"output":4.5,"input_audio":3,"output_audio":12}},"gemini-2.5-pro-preview-tts":{"id":"gemini-2.5-pro-preview-tts","name":"Gemini 2.5 Pro Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"veo-3.1-generate-preview":{"id":"veo-3.1-generate-preview","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192},"status":"beta"},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Legacy model retained for compatibility with older integrations","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"veo-3.1-fast-generate-preview":{"id":"veo-3.1-fast-generate-preview","name":"Veo 3.1 fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"gemini-2.5-flash-preview-tts":{"id":"gemini-2.5-flash-preview-tts","name":"Gemini 2.5 Flash Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":60}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":120}},"gemini-3.1-flash-tts-preview":{"id":"gemini-3.1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"veo-3.1-lite-generate-preview":{"id":"veo-3.1-lite-generate-preview","name":"Veo 3.1 lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1},"cost":{"input":0.2,"output":0,"input_audio":6.5}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.5-live-translate-preview":{"id":"gemini-3.5-live-translate-preview","name":"Gemini 3.5 Live Translate Preview","description":"Low-latency audio-to-audio model for real-time speech translation across 70+ languages","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["audio"],"output":["audio","text"]},"open_weights":false,"limit":{"context":16384,"output":32768},"cost":{"input":3.5,"output":21,"input_audio":3.5,"output_audio":21}},"lyria-3-pro-preview":{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Music generation model for full-length songs from text or images with vocals and structure","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":60}},"gemini-omni-flash-preview":{"id":"gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Video generation and editing model for fast, conversational text- and image-to-video workflows","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1.5,"output":17.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}}}},"baseten":{"id":"baseten","env":["BASETEN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.baseten.co/v1","name":"Baseten","doc":"https://docs.baseten.co/inference/model-apis/overview","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":131000},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/Nemotron-120B-A12B":{"id":"nvidia/Nemotron-120B-A12B","name":"Nemotron Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.3,"output":0.75,"cache_read":0.06}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"zai-org/GLM-5.2-Fast":{"id":"zai-org/GLM-5.2-Fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.3}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.95,"output":3.15,"cache_read":0.2}},"zai-org/GLM-5.3-Fast":{"id":"zai-org/GLM-5.3-Fast","name":"GLM 5.3 Fast","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1,"output":4.05}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204000,"output":204000},"status":"deprecated","cost":{"input":0.3,"output":1.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128072,"output":128072},"cost":{"input":0.1,"output":0.5}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-30","last_updated":"2026-02-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.6,"output":3,"cache_read":0.12}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15}}}},"vercel":{"id":"vercel","env":["AI_GATEWAY_API_KEY"],"npm":"@ai-sdk/gateway","name":"Vercel AI Gateway","doc":"https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway","models":{"voyage/voyage-code-3":{"id":"voyage/voyage-code-3","name":"voyage-code-3","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-04","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3.5":{"id":"voyage/voyage-3.5","name":"voyage-3.5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3.5-lite":{"id":"voyage/voyage-3.5-lite","name":"voyage-3.5-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3-large":{"id":"voyage/voyage-3-large","name":"voyage-3-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-07","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-code-2":{"id":"voyage/voyage-code-2","name":"voyage-code-2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4":{"id":"voyage/voyage-4","name":"voyage-4","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/voyage-finance-2":{"id":"voyage/voyage-finance-2","name":"voyage-finance-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-06-03","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/rerank-2.5":{"id":"voyage/rerank-2.5","name":"Voyage Rerank 2.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-law-2":{"id":"voyage/voyage-law-2","name":"voyage-law-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-15","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4-large":{"id":"voyage/voyage-4-large","name":"voyage-4-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/rerank-2.5-lite":{"id":"voyage/rerank-2.5-lite","name":"Voyage Rerank 2.5 Lite","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-4-lite":{"id":"voyage/voyage-4-lite","name":"voyage-4-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"poolside/laguna-s-2.1-free":{"id":"poolside/laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"klingai/kling-v2.5-turbo-i2v":{"id":"klingai/kling-v2.5-turbo-i2v","name":"Kling v2.5 Turbo Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-motion-control":{"id":"klingai/kling-v3.0-motion-control","name":"Kling v3.0 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-t2v":{"id":"klingai/kling-v2.6-t2v","name":"Kling v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.5-turbo-t2v":{"id":"klingai/kling-v2.5-turbo-t2v","name":"Kling v2.5 Turbo Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-t2v":{"id":"klingai/kling-v3.0-t2v","name":"Kling v3.0 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-i2v":{"id":"klingai/kling-v2.6-i2v","name":"Kling v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-motion-control":{"id":"klingai/kling-v2.6-motion-control","name":"Kling v2.6 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-i2v":{"id":"klingai/kling-v3.0-i2v","name":"Kling v3.0 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"StepFun 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"stepfun/step-5-preview":{"id":"stepfun/step-5-preview","name":"Step 5 Preview","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"interfaze/interfaze-beta":{"id":"interfaze/interfaze-beta","name":"Interfaze Beta","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"release_date":"2025-10-07","last_updated":"2026-04-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.5,"output":3.5}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo M2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131100},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-23","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-h3":{"id":"minimax/minimax-h3","name":"MiniMax H3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 High Speed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"Minimax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 High Speed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-h3-max":{"id":"minimax/minimax-h3-max","name":"MiniMax H3 Max","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen3-vl-thinking":{"id":"alibaba/qwen3-vl-thinking","name":"Qwen3 VL Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen-3-235b":{"id":"alibaba/qwen-3-235b","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.88}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"tiers":[{"input":1.8,"output":9,"cache_read":0.36,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}},{"input":6,"output":60,"cache_read":1.2,"tier":{"type":"context","size":256001}}]}},"alibaba/qwen3-next-80b-a3b-thinking":{"id":"alibaba/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.15,"output":1.2}},"alibaba/qwen3-coder-30b-a3b":{"id":"alibaba/qwen3-coder-30b-a3b","name":"Qwen 3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"alibaba/qwen3-next-80b-a3b-instruct":{"id":"alibaba/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.15,"output":1.2}},"alibaba/wan-v3.0-video":{"id":"alibaba/wan-v3.0-video","name":"Wan v3.0 Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-23","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen 3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"alibaba/wan-v2.7-r2v":{"id":"alibaba/wan-v2.7-r2v","name":"Wan v2.7 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-27b":{"id":"alibaba/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"alibaba/wan-v2.6-t2v":{"id":"alibaba/wan-v2.6-t2v","name":"Wan v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v3.0-video-prime":{"id":"alibaba/wan-v3.0-video-prime","name":"Wan v3.0 Video Prime","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen-3-14b":{"id":"alibaba/qwen-3-14b","name":"Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.24}},"alibaba/qwen3-vl-instruct":{"id":"alibaba/qwen3-vl-instruct","name":"Qwen3 VL Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3-235b-a22b-thinking":{"id":"alibaba/qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen3.5-flash":{"id":"alibaba/qwen3.5-flash","name":"Qwen 3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"alibaba/qwen3-coder":{"id":"alibaba/qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.3,"tiers":[{"input":2.7,"output":13.5,"cache_read":0.54,"tier":{"type":"context","size":32001}},{"input":4.5,"output":22.5,"cache_read":0.9,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3-coder-next":{"id":"alibaba/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.2}},"alibaba/qwen3-embedding-4b":{"id":"alibaba/qwen3-embedding-4b","name":"Qwen3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3-max-preview":{"id":"alibaba/qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-05","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3-embedding-8b":{"id":"alibaba/qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3.6-27b":{"id":"alibaba/qwen3.6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3.6}},"alibaba/wan-v2.6-r2v":{"id":"alibaba/wan-v2.6-r2v","name":"Wan v2.6 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen 3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"cache_write":0.125,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"cache_write":0.25,"tier":{"type":"context","size":256000}}]}},"alibaba/qwen3.8-omni-flash":{"id":"alibaba/qwen3.8-omni-flash","name":"Qwen 3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"alibaba/qwen3-max-thinking":{"id":"alibaba/qwen3-max-thinking","name":"Qwen 3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-23","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3.8-max-0902":{"id":"alibaba/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/wan-v2.6-i2v-flash":{"id":"alibaba/wan-v2.6-i2v-flash","name":"Wan v2.6 Image-to-Video Flash","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-2.4t-a95b":{"id":"alibaba/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/wan-v2.6-r2v-flash":{"id":"alibaba/wan-v2.6-r2v-flash","name":"Wan v2.6 Reference-to-Video Flash","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v2.5-t2v-preview":{"id":"alibaba/wan-v2.5-t2v-preview","name":"Wan v2.5 Text-to-Video Preview","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen-3-30b":{"id":"alibaba/qwen-3-30b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/wan-v2.6-i2v":{"id":"alibaba/wan-v2.6-i2v","name":"Wan v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen-3.6-max-preview":{"id":"alibaba/qwen-3.6-max-preview","name":"Qwen 3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":240000,"output":64000},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625,"tiers":[{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":128000}}]}},"alibaba/qwen3-embedding-0.6b":{"id":"alibaba/qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3-vl-235b-a22b-instruct":{"id":"alibaba/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5}}},"alibaba/qwen-3-32b":{"id":"alibaba/qwen-3-32b","name":"Qwen 3.32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.16,"output":0.64}},"alibaba/wan-v2.7-t2v":{"id":"alibaba/wan-v2.7-t2v","name":"Wan v2.7 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.5-plus":{"id":"alibaba/qwen3.5-plus","name":"Qwen 3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.5,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":256001}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nvidia Nemotron Nano 9B V2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.06,"output":0.23}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nvidia Nemotron Nano 12B V2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.6}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.15,"output":0.65}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"spacexai/grok-voice-think-fast-2.0":{"id":"spacexai/grok-voice-think-fast-2.0","name":"Grok Voice Think Fast 2.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-reasoning":{"id":"spacexai/grok-4.20-reasoning","name":"Grok 4.20 Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.20-multi-agent":{"id":"spacexai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.3":{"id":"spacexai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.20-reasoning-beta":{"id":"spacexai/grok-4.20-reasoning-beta","name":"Grok 4.20 Beta Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-imagine-image":{"id":"spacexai/grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-tts":{"id":"spacexai/grok-tts","name":"Grok TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-non-reasoning":{"id":"spacexai/grok-4.20-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-imagine-video":{"id":"spacexai/grok-imagine-video","name":"Grok Imagine","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.5":{"id":"spacexai/grok-4.5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"spacexai/grok-build-0.1":{"id":"spacexai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"spacexai/grok-4.20-multi-agent-beta":{"id":"spacexai/grok-4.20-multi-agent-beta","name":"Grok 4.20 Multi Agent Beta","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.20-non-reasoning-beta":{"id":"spacexai/grok-4.20-non-reasoning-beta","name":"Grok 4.20 Beta Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.4,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.1-fast-non-reasoning":{"id":"spacexai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-4.1-fast-reasoning":{"id":"spacexai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-imagine-video-1.5":{"id":"spacexai/grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-imagine-image-2.0":{"id":"spacexai/grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.6":{"id":"spacexai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"spacexai/grok-stt":{"id":"spacexai/grok-stt","name":"Grok STT","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-voice-think-fast-1.0":{"id":"spacexai/grok-voice-think-fast-1.0","name":"Grok Voice Think Fast 1.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.7":{"id":"spacexai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":1.2,"output":3.6,"cache_read":0.3,"tiers":[{"input":2.4,"output":7.2,"cache_read":0.6,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.4,"output":7.2,"cache_read":0.6}}},"prodia/flux-fast-schnell":{"id":"prodia/flux-fast-schnell","name":"Flux Schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5-fast":{"id":"anthropic/claude-opus-5-fast","name":"Claude Opus 5 (Fast)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude Haiku 3","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.8-fast":{"id":"anthropic/claude-opus-4.8-fast","name":"Claude Opus 4.8 (Fast)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/gemini-3.5-transcribe-live":{"id":"google/gemini-3.5-transcribe-live","name":"Gemini 3.5 Transcribe Live","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana (Gemini 2.5 Flash Image)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-3.5-transcribe":{"id":"google/gemini-3.5-transcribe","name":"Gemini 3.5 Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":12}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/veo-3.1-fast-generate-001":{"id":"google/veo-3.1-fast-generate-001","name":"Veo 3.1 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.8-live":{"id":"google/gemini-3.8-live","name":"Gemini 3.8 Live","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-15","last_updated":"2026-09-15","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.75,"output":4.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/text-embedding-005":{"id":"google/text-embedding-005","name":"Text Embedding 005","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-01","last_updated":"2024-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.8-live-extended-thinking":{"id":"google/gemini-3.8-live-extended-thinking","name":"Gemini 3.8 Live Extended Thinking","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-15","last_updated":"2026-09-15","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.75,"output":4.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/veo-3.0-generate-001":{"id":"google/veo-3.0-generate-001","name":"Veo 3.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/veo-3.1-generate-001":{"id":"google/veo-3.1-generate-001","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemini-embedding-2":{"id":"google/gemini-embedding-2","name":"Gemini Embedding 2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/veo-3.0-fast-generate-001":{"id":"google/veo-3.0-fast-generate-001","name":"Veo 3.0 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/text-multilingual-embedding-002":{"id":"google/text-multilingual-embedding-002","name":"Text Multilingual Embedding 002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-01","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Gemini 3.1 Flash Image Preview (Nano Banana 2)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-omni-flash-preview":{"id":"google/gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":57920},"cost":{"input":1.5,"output":9}},"google/veo-3.1-lite-generate-001":{"id":"google/veo-3.1-lite-generate-001","name":"Veo 3.1 Lite Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"bfl/flux-kontext-max":{"id":"bfl/flux-kontext-max","name":"FLUX.1 Kontext Max","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-2-pro":{"id":"bfl/flux-2-pro","name":"FLUX.2 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-2-max":{"id":"bfl/flux-2-max","name":"FLUX.2 [max]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-pro-1.1-ultra":{"id":"bfl/flux-pro-1.1-ultra","name":"FLUX1.1 [pro] Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-kontext-pro":{"id":"bfl/flux-kontext-pro","name":"FLUX.1 Kontext Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-3-video":{"id":"bfl/flux-3-video","name":"Flux 3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-pro-1.0-fill":{"id":"bfl/flux-pro-1.0-fill","name":"FLUX.1 Fill [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-2-flex":{"id":"bfl/flux-2-flex","name":"FLUX.2 [flex]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-klein-9b":{"id":"bfl/flux-2-klein-9b","name":"FLUX.2 [klein] 9B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-klein-4b":{"id":"bfl/flux-2-klein-4b","name":"FLUX.2 [klein] 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-pro-1.1":{"id":"bfl/flux-pro-1.1","name":"FLUX1.1 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-02","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"mixedbread/toast-1":{"id":"mixedbread/toast-1","name":"Toast 1","description":"Specialized search model for knowledge-intensive questions, multi-step retrieval, and evidence synthesis","attachment":false,"reasoning":false,"tool_call":true,"release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":4000},"cost":{"input":0.3,"output":0.72,"cache_read":0.036}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/llama-3.1-8b":{"id":"meta/llama-3.1-8b","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.22,"output":0.22}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"muse","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"meta/llama-3.1-70b":{"id":"meta/llama-3.1-70b","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.72,"output":0.72}},"meta/muse-image-1.0":{"id":"meta/muse-image-1.0","name":"Muse Image 1.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"muse","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"meta/llama-3.3-70b":{"id":"meta/llama-3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-scout":{"id":"meta/llama-4-scout","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-maverick":{"id":"meta/llama-4-maverick","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"quiverai/arrow-2-telos":{"id":"quiverai/arrow-2-telos","name":"Arrow 2 Telos","description":"High-fidelity SVG generation model for complex vector work and long-context refinement","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-16","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"quiverai/arrow-1.1":{"id":"quiverai/arrow-1.1","name":"Arrow 1.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"quiverai/arrow-2":{"id":"quiverai/arrow-2","name":"Arrow 2","description":"Fast SVG generation model for creation, vectorization, editing, and animation","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-16","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"bytedance/seedance-2.0-mini":{"id":"bytedance/seedance-2.0-mini","name":"Seedance 2.0 Mini","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro-fast":{"id":"bytedance/seedance-v1.0-pro-fast","name":"Seedance v1.0 Pro Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-31","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro":{"id":"bytedance/seedance-v1.0-pro","name":"Seedance v1.0 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-11","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.5-pro":{"id":"bytedance/seedance-v1.5-pro","name":"Seedance v1.5 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-4.5":{"id":"bytedance/seedream-4.5","name":"Seedream 4.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-11-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seed-1.8":{"id":"bytedance/seed-1.8","name":"Seed 1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.25,"output":2,"cache_read":0.05,"tiers":[{"input":0.5,"output":4,"cache_read":0.05,"tier":{"type":"context","size":128001}}]}},"bytedance/seed-2.1-turbo":{"id":"bytedance/seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.5,"cache_read":0.1}},"bytedance/seed-1.6":{"id":"bytedance/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.25,"output":2,"cache_read":0.05,"tiers":[{"input":0.5,"output":4,"cache_read":0.05,"tier":{"type":"context","size":128001}}]}},"bytedance/seedance-2.0-fast":{"id":"bytedance/seedance-2.0-fast","name":"Seedance 2.0 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-4.0":{"id":"bytedance/seedream-4.0","name":"Seedream 4.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-lite":{"id":"bytedance/seedream-5.0-lite","name":"Seedream 5.0 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.5":{"id":"bytedance/seedance-2.5","name":"Seedance 2.5","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.0":{"id":"bytedance/seedance-2.0","name":"Seedance 2.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-pro":{"id":"bytedance/seedream-5.0-pro","name":"Seedream 5.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-11","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"inception/mercury-coder-small":{"id":"inception/mercury-coder-small","name":"Mercury Coder Small Beta","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"mercury","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-02-26","last_updated":"2025-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":16384},"cost":{"input":0.25,"output":1}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.25,"output":0.75,"cache_read":0.024999999999999998}},"fish-audio/transcribe-1":{"id":"fish-audio/transcribe-1","name":"Transcribe-1","description":"Speech transcription model for accurate audio-to-text and captioning workflows","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2.1-pro":{"id":"fish-audio/s2.1-pro","name":"S2.1 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-28","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2-pro":{"id":"fish-audio/s2-pro","name":"S2 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s1":{"id":"fish-audio/s1","name":"S1","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Fugu Ultra v2","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/namazu":{"id":"sakana/namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.066}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"deepseek/deepseek-v3.2-thinking":{"id":"deepseek/deepseek-v3.2-thinking","name":"DeepSeek V3.2 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":128000},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"amazon/nova-2-lite":{"id":"amazon/nova-2-lite","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2024-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}},"amazon/titan-embed-text-v2":{"id":"amazon/titan-embed-text-v2","name":"Titan Text Embeddings V2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"titan-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-30","last_updated":"2024-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"amazon/nova-pro":{"id":"amazon/nova-pro","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"amazon/nova-lite":{"id":"amazon/nova-lite","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"amazon/nova-micro":{"id":"amazon/nova-micro","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.021,"output":0.063,"cache_read":0.0042}},"inclusionai/ling-3.0-flash-vl-free":{"id":"inclusionai/ling-3.0-flash-vl-free","name":"Ling 3.0 Flash VL (Free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante":{"id":"inclusionai/ling-3.0-flash-sante","name":"Ling 3.0 Flash Sante","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante-free":{"id":"inclusionai/ling-3.0-flash-sante-free","name":"Ling 3.0 Flash Sante (Free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-fin-free":{"id":"inclusionai/ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin (Free)","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"openai/gpt-5-fast":{"id":"openai/gpt-5-fast","name":"GPT-5 (Fast)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":128000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":1.25,"output":5}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.5-fast":{"id":"openai/gpt-5.5-fast","name":"GPT 5.5 (Fast)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":12.5,"output":75,"cache_read":1.25}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.4-mini-fast":{"id":"openai/gpt-5.4-mini-fast","name":"GPT 5.4 Mini (Fast)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT-Realtime-1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":16,"cache_read":0.4}},"openai/gpt-5-mini-fast":{"id":"openai/gpt-5-mini-fast","name":"GPT-5 mini (Fast)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.45,"output":3.6,"cache_read":0.045}},"openai/tts-1":{"id":"openai/tts-1","name":"TTS-1","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-6-astra-fast":{"id":"openai/gpt-6-astra-fast","name":"GPT-6 Astra (Fast)","description":"Fast variant of GPT-6 Astra for low-latency assistance and high-volume workloads.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25,"tiers":[{"input":40,"output":150,"cache_read":4,"cache_write":50,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":40,"output":150,"cache_read":4,"cache_write":50}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT 5.2 ","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-realtime-2":{"id":"openai/gpt-realtime-2","name":"gpt-realtime-2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":122880,"output":8192},"cost":{"input":0.03,"output":0.14}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2.5,"output":10}},"openai/o4-mini-fast":{"id":"openai/o4-mini-fast","name":"o4-mini (Fast)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.07,"output":0.2}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.15,"output":0.6}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-4.1-fast":{"id":"openai/gpt-4.1-fast","name":"GPT-4.1 (Fast)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-4o-mini-fast":{"id":"openai/gpt-4o-mini-fast","name":"GPT-4o mini (Fast)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"openai/gpt-5.6-luna-fast":{"id":"openai/gpt-5.6-luna-fast","name":"GPT 5.6 Luna (Fast)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":0.8,"output":3.6,"cache_read":0.08,"cache_write":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.8,"output":3.6,"cache_read":0.08,"cache_write":1}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"openai/gpt-5.4-fast":{"id":"openai/gpt-5.4-fast","name":"GPT 5.4 (Fast)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.1-thinking-fast":{"id":"openai/gpt-5.1-thinking-fast","name":"GPT 5.1 Thinking (Fast)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/text-embedding-ada-002":{"id":"openai/text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT Image 1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.1-thinking":{"id":"openai/gpt-5.1-thinking","name":"GPT 5.1 Thinking","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-11-12","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT 5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/tts-1-hd":{"id":"openai/tts-1-hd","name":"TTS-1 HD","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/o3-fast":{"id":"openai/o3-fast","name":"o3 (Fast)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT Image 1 Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":8,"cache_read":0.2}},"openai/gpt-live-1":{"id":"openai/gpt-live-1","name":"GPT-Live 1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.6-terra-fast":{"id":"openai/gpt-5.6-terra-fast","name":"GPT 5.6 Terra (Fast)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":36,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":36,"cache_read":0.8,"cache_write":10}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-image-2.5-sunburst":{"id":"openai/gpt-image-2.5-sunburst","name":"GPT Image 2.5 Sunburst","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-4.1-nano-fast":{"id":"openai/gpt-4.1-nano-fast","name":"GPT-4.1 nano (Fast)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"openai/gpt-realtime-mini":{"id":"openai/gpt-realtime-mini","name":"GPT-Realtime mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-4.1-mini-fast":{"id":"openai/gpt-4.1-mini-fast","name":"GPT-4.1 mini (Fast)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.7,"output":2.8,"cache_read":0.175}},"openai/gpt-realtime-whisper":{"id":"openai/gpt-realtime-whisper","name":"gpt-realtime-whisper","description":"Streaming speech-to-text model for low-latency transcript deltas from live audio","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":12289,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.3-codex-fast":{"id":"openai/gpt-5.3-codex-fast","name":"GPT 5.3 Codex (Fast)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/text-embedding-3-small":{"id":"openai/text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT 5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/text-embedding-3-large":{"id":"openai/text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-image-2.5-flare":{"id":"openai/gpt-image-2.5-flare","name":"GPT Image 2.5 Flare","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.2-fast":{"id":"openai/gpt-5.2-fast","name":"GPT 5.2 (Fast)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-sol-fast":{"id":"openai/gpt-5.6-sol-fast","name":"GPT 5.6 Sol (Fast)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10,"tiers":[{"input":16,"output":60,"cache_read":1.6,"cache_write":20,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":16,"output":60,"cache_read":1.6,"cache_write":20}}},"openai/whisper-1":{"id":"openai/whisper-1","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-09-21","last_updated":"2022-09-21","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-fast":{"id":"openai/gpt-4o-fast","name":"GPT-4o (Fast)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":4.25,"output":17,"cache_read":2.125}},"openai/gpt-realtime-2.1":{"id":"openai/gpt-realtime-2.1","name":"gpt-realtime-2.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3 Pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":216144,"output":216144},"cost":{"input":0.47,"output":2,"cache_read":0.141}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k3-fast":{"id":"moonshotai/kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"cohere/rerank-v4-pro":{"id":"cohere/rerank-v4-pro","name":"Cohere Rerank 4 Pro","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/rerank-v4-fast":{"id":"cohere/rerank-v4-fast","name":"Cohere Rerank 4 Fast","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/embed-v4.0":{"id":"cohere/embed-v4.0","name":"Embed v4.0","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":1536}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/rerank-v3.5":{"id":"cohere/rerank-v3.5","name":"Cohere Rerank 3.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":80000},"cost":{"input":0.25,"output":0.8999999999999999}},"tencent/hy-mt2-lite":{"id":"tencent/hy-mt2-lite","name":"Tencent Hy-MT2-Lite","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.044,"output":0.177}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-plus":{"id":"tencent/hy-mt2-plus","name":"Tencent Hy-MT2-Plus","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-pro":{"id":"tencent/hy-mt2-pro","name":"Tencent Hy-MT2-Pro","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":120000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"zai/glm-5.3-flashx":{"id":"zai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075}},"zai/glm-5.3-fast":{"id":"zai/glm-5.3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai/glm-5.2-fast":{"id":"zai/glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":66000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":64000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"output":131100},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.07,"output":0.4}},"mistral/mistral-embed":{"id":"mistral/mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/mistral-large-3":{"id":"mistral/mistral-large-3","name":"Mistral Large 3","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/mistral-nemo":{"id":"mistral/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-07-18","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":60288,"output":16000},"cost":{"input":0.04,"output":0.17}},"mistral/codestral-embed":{"id":"mistral/codestral-embed","name":"Codestral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"codestral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/mistral-small":{"id":"mistral/mistral-small","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2024-09-17","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/ministral-14b":{"id":"mistral/ministral-14b","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistral/mistral-medium-3.5":{"id":"mistral/mistral-medium-3.5","name":"Mistral Medium Latest","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-05-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/codestral":{"id":"mistral/codestral","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/ministral-8b":{"id":"mistral/ministral-8b","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"mistral/ministral-3b":{"id":"mistral/ministral-3b","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"perplexity/pplx-embed-v1-4b":{"id":"perplexity/pplx-embed-v1-4b","name":"Embed v1 4b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/pplx-embed-v1-0.6b":{"id":"perplexity/pplx-embed-v1-0.6b","name":"Embed v1 0.6b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"v0","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000}},"recraft/recraft-v4-pro":{"id":"recraft/recraft-v4-pro","name":"Recraft V4 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility":{"id":"recraft/recraft-v4.1-utility","name":"Recraft V4.1 Utility","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility-pro":{"id":"recraft/recraft-v4.1-utility-pro","name":"Recraft V4.1 Utility Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v2":{"id":"recraft/recraft-v2","name":"Recraft V2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v3":{"id":"recraft/recraft-v3","name":"Recraft V3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-30","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v4.1-pro":{"id":"recraft/recraft-v4.1-pro","name":"Recraft V4.1 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4":{"id":"recraft/recraft-v4","name":"Recraft V4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1":{"id":"recraft/recraft-v4.1","name":"Recraft V4.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}}}},"qvac":{"id":"qvac","env":["QVAC_API_KEY"],"npm":"@qvac/ai-sdk-provider","name":"QVAC","doc":"https://www.npmjs.com/package/@qvac/ai-sdk-provider","models":{"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.5-0.8b":{"id":"qwen3.5-0.8b","name":"Qwen3.5 0.8B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.5-4b":{"id":"qwen3.5-4b","name":"Qwen3.5 4B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"Qwen3.5 2B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"wandb":{"id":"wandb","env":["WANDB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.wandb.ai/v1","name":"CoreWeave","doc":"https://docs.wandb.ai/inference","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4-Flash is an MoE model with 1M context length great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4-Flash-0731 is an MoE model great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"A large hybrid model that supports both thinking and non-thinking modes via prompt templates.","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":161000,"output":161000},"cost":{"input":0.55,"output":1.65,"cache_read":0.55}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4-Pro-0813 is a 1.6T-parameter MoE model excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.31,"output":3.96,"cache_read":0.044}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4-Pro is a 1.6T-parameter MoE model with 49B active parameters excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.15,"output":2.55,"cache_read":0.2}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron 3 Ultra","description":"Nemotron 3 Ultra is a powerful MoE model designed for long-running agents across coding, deep research, and enterprise automation.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.15,"cache_read":0.1}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B","name":"Nemotron 3.5 Lightning","description":"Nemotron 3.5 Lightning is an MoE model built for fast, reliable agentic tasks across use cases such as financial services, cybersecurity, telecom, and retail.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.2,"cache_read":0.04}},"OpenPipe/Qwen3-14B-Instruct":{"id":"OpenPipe/Qwen3-14B-Instruct","name":"Qwen3 14B Instruct","description":"An efficient multilingual, dense, instruction-tuned model, optimized by OpenPipe for building agents with finetuning.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.05,"output":0.22,"cache_read":0.05}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B","description":"Gemma 4 31B Dense is designed for advanced reasoning, agentic workflows, and longer context and is natively trained on 140+ languages.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.34,"cache_read":0.1}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"GLM-5.2 is a Mixture-of-Experts language model featuring 40 billion activated parameters and a total of 744 billion parameters.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.76,"output":2.42,"cache_read":0.14}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM-5.3-Flash is a natively multimodal model with 320B total parameters and 18B active parameters.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.05}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B MoE instruction-tuned model with enhanced reasoning, coding, and long-context understanding.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3,"cache_read":0.1}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen3.8-27B is a dense multimodal model suited for coding, research, vision, and long-running agent tasks.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5-35B-A3B","description":"Qwen3.5-35B-A3B is an open-weights multimodal MoE model built for efficient, high-throughput inference across chat, reasoning, and agentic tasks.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen3.6-27B is a 27B dense multimodal model with 262K context built for flagship-level agentic coding.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6,"cache_read":0.12}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen3.6-35B-A3B is an MoE multimodal model with 262K context optimized for agentic coding workflows.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"ibm-granite/granite-4.1-8b":{"id":"ibm-granite/granite-4.1-8b","name":"Granite 4.1 8B","description":"Granite 4.1 8B is a long-context instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Granite 4.2 8B is an instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-24","last_updated":"2026-08-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax M3","description":"MiniMax M3 is a multimodal MoE model with 23B active parameters optimized for coding and agentic workflows.","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.23,"output":0.96,"cache_read":0.05}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.22,"output":0.22,"cache_read":0.22}},"meta-llama/Llama-3.1-70B-Instruct":{"id":"meta-llama/Llama-3.1-70B-Instruct","name":"Llama 3.1 70B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.8,"output":0.8,"cache_read":0.8}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B","description":"Multilingual model excelling in conversational tasks, detailed instruction-following, and coding.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.71,"output":0.71,"cache_read":0.71}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"gpt-oss-20b","description":"Lower latency Mixture-of-Experts model trained on OpenAI's Harmony response format with reasoning capabilities.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.13,"cache_read":0.03}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Efficient Mixture-of-Experts model designed for high-reasoning, agentic and general-purpose use cases.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi K2.7 Code is a 1T-parameter MoE model with 32B active parameters purpose-built for long-horizon agentic coding and software engineering.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":3.5,"cache_read":0.15}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.65,"output":3.41,"cache_read":0.15}},"JetBrains/Mellum2-12B-A2.5B-Instruct":{"id":"JetBrains/Mellum2-12B-A2.5B-Instruct","name":"Mellum2 12B A2.5B","description":"Mellum2-12B-A2.5B-Instruct is a fast MoE model with 131K context built for coding, tool use, and low-latency AI workflows.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}}}},"friendli":{"id":"friendli","env":["FRIENDLI_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.friendli.ai/serverless/v1","name":"Friendli","doc":"https://friendli.ai/docs/guides/serverless_endpoints/introduction","models":{"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}}}},"tokenrouter":{"id":"tokenrouter","env":["TOKENROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenrouter.com/v1","name":"TokenRouter","doc":"https://www.tokenrouter.com/docs/tokenrouter-feature-guide/","models":{"z-ai/glm-5.3-free":{"id":"z-ai/glm-5.3-free","name":"GLM-5.3 (free)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"thinkingmachines":{"id":"thinkingmachines","env":["TINKER_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://tinker.thinkingmachines.dev/services/tinker-prod/anthropic/api/v1","name":"Thinking Machines","doc":"https://tinker-docs.thinkingmachines.ai/tinker/compatible-apis/anthropic/","models":{"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"thinkingmachines/Inkling:peft:262144":{"id":"thinkingmachines/Inkling:peft:262144","name":"Inkling (256K)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}}}},"standardcompute":{"id":"standardcompute","env":["STANDARDCOMPUTE_API_KEY"],"npm":"@openrouter/ai-sdk-provider","api":"https://api.stdcmpt.com/v1","name":"Standard Compute","doc":"https://standardcompute.com/models","models":{"standardcompute":{"id":"standardcompute","name":"Standard Compute","description":"Flat-rate smart-routing gateway: one model id, each request routed across a curated catalog of 1M-context models (DeepSeek, GLM, MiniMax, Qwen, GPT-5.6, Claude 5, Gemini 2.5, Kimi) or pinned to a user-selected model","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":24576},"cost":{"input":0,"output":0}}}},"tensorx":{"id":"tensorx","env":["TENSORX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tensorx.ai/v1","name":"TensorX","doc":"https://docs.tensorx.ai/","models":{"qwen/qwen3.8-flash-next":{"id":"qwen/qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.2,"cache_read":0.0375,"cache_write":0.1875}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":2.4,"cache_read":0.1}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B-A22B-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":262144},"cost":{"input":0.072,"output":0.464,"cache_read":0.018,"cache_write":0.09}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":3.5,"cache_read":0.125,"cache_write":0.625}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.075,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":2,"output":4,"cache_read":0.5}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.06}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.5,"output":1.5,"cache_read":0.13}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1-0528","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":8192},"cost":{"input":0.66,"output":2.6,"cache_read":0.165,"cache_write":0.825}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.3,"output":0.5,"cache_read":0.075,"cache_write":0.375}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.75,"output":3.5,"cache_read":0.4375,"cache_write":2.185}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1,"output":4,"cache_read":0.25,"cache_write":1.25}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.3125}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125,"cache_write":0.625}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.5,"output":4.5,"cache_read":0.375}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1,"output":3.2,"cache_read":0.25,"cache_write":1.25}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.4,"output":4.4,"cache_read":0.35,"cache_write":1.75}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":1.75,"output":4.5,"cache_read":0.44}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}}}},"meta":{"id":"meta","env":["META_MODEL_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.meta.ai/v1","name":"Meta","doc":"https://dev.meta.ai/docs","models":{"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}}}},"venice":{"id":"venice","env":["VENICE_API_KEY"],"npm":"venice-ai-sdk-provider","name":"Venice AI","doc":"https://docs.venice.ai","models":{"google-gemma-3-27b-it":{"id":"google-gemma-3-27b-it","name":"Google Gemma 3 27B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-04","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.12,"output":0.2}},"zai-org-glm-5-2":{"id":"zai-org-glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.6,"output":18,"cache_read":0.36,"cache_write":4.5}},"deepseek-v4-flash-0731-fast":{"id":"deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731 Fast","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-09","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.35,"output":0.7,"cache_read":0.0875}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen 3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.15}},"openai-gpt-55-pro":{"id":"openai-gpt-55-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225}},"zai-org-glm-4.7-flash":{"id":"zai-org-glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"mistral-small-3-2-24b-instruct":{"id":"mistral-small-3-2-24b-instruct","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":4.95,"cache_read":0.165}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.175,"output":0.35,"cache_read":0.035}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"venice-uncensored-1-2":{"id":"venice-uncensored-1-2","name":"Venice Uncensored 1.2","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen 3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.45,"output":3.5}},"gemma-4-uncensored":{"id":"gemma-4-uncensored","name":"Gemma 4 Uncensored","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-13","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1625,"output":0.5}},"zai-org-glm-5-1":{"id":"zai-org-glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":80000},"cost":{"input":1.54,"output":4.84,"cache_read":0.286}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus Uncensored","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-06","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.625,"output":3.75,"cache_read":0.0625,"cache_write":0.78,"tiers":[{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78}}},"openai-gpt-55":{"id":"openai-gpt-55","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":6.25,"output":37.5,"cache_read":0.625,"tiers":[{"input":12.5,"output":56.25,"cache_read":1.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":12.5,"output":56.25,"cache_read":1.25}}},"claude-opus-5-fast":{"id":"claude-opus-5-fast","name":"Claude Opus 5 Fast","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"minimax-m25":{"id":"minimax-m25","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.27,"output":0.95,"cache_read":0.03}},"aion-labs-aion-3-0-mini":{"id":"aion-labs-aion-3-0-mini","name":"Aion 3.0 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.875,"output":1.75,"cache_read":0.225}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.55,"output":9.45,"cache_read":0.155,"cache_write":0.086}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"qwen-3-7-max":{"id":"qwen-3-7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.7,"output":8.05,"cache_read":0.27,"cache_write":3.35}},"qwen-3-8-max":{"id":"qwen-3-8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-22","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125,"cache_write":3.125}},"openai-gpt-54":{"id":"openai-gpt-54","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":3.13,"output":18.8,"cache_read":0.313}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.375,"output":1.5,"cache_read":0.0075}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-05","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"z-ai-glm-5-turbo":{"id":"z-ai-glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai-org-glm-4.6":{"id":"zai-org-glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2024-04-01","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-06","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":32768},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash 0423","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.138,"output":0.275,"cache_read":0.028}},"zai-org-glm-5":{"id":"zai-org-glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"openai-gpt-56-terra":{"id":"openai-gpt-56-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"kimi-k3-fast-api":{"id":"kimi-k3-fast-api","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"qwen-3-8-27b":{"id":"qwen-3-8-27b","name":"Qwen 3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-17","last_updated":"2026-08-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.2}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"olafangensan-glm-4.7-flash-heretic":{"id":"olafangensan-glm-4.7-flash-heretic","name":"GLM 4.7 Flash Heretic","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":24000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-13","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"aion-labs-aion-3-0":{"id":"aion-labs-aion-3-0","name":"Aion 3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":3.75,"output":7.5,"cache_read":0.9375}},"grok-4-7":{"id":"grok-4-7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-16","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":200000},"cost":{"input":2.27,"output":6.8,"cache_read":0.57,"tiers":[{"input":4.53,"output":13.6,"cache_read":1.13,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":1.13}}},"llama-3.2-3b":{"id":"llama-3.2-3b","name":"Llama 3.2 3B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-10-03","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.6}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen 3.5 397B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.75,"output":4.5}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-28","last_updated":"2026-07-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.625,"output":3.125,"cache_read":0.125}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-08-29","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":0.3,"cache_write":15}},"openai-gpt-54-pro":{"id":"openai-gpt-54-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225,"tiers":[{"input":75,"output":337.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":75,"output":337.5}}},"hermes-3-llama-3.1-405b":{"id":"hermes-3-llama-3.1-405b","name":"Hermes 3 Llama 3.1 405b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"hermes","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-09-25","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":3}},"venice-uncensored-role-play":{"id":"venice-uncensored-role-play","name":"Venice Role Play Uncensored","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":2}},"mercury-2-5":{"id":"mercury-2-5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-08","last_updated":"2026-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04999999999999999,"output":0.18749999999999994,"cache_read":0.004999999999999999}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"openai-gpt-6-astra-pro":{"id":"openai-gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-05","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":12.5,"output":62.5,"cache_read":1.25,"cache_write":15.625,"tiers":[{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25}}},"openai-gpt-54-mini":{"id":"openai-gpt-54-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.9375,"output":5.625,"cache_read":0.09375}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen 3.6 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.1,"output":1}},"minimax-m3-preview":{"id":"minimax-m3-preview","name":"MiniMax M3 Preview","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-12","last_updated":"2026-06-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"kimi-k2-5":{"id":"kimi-k2-5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-04","release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.56,"output":3.5,"cache_read":0.22}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.375,"output":3.125,"cache_read":0.0375}},"google-gemma-4-26b-a4b-it":{"id":"google-gemma-4-26b-a4b-it","name":"Google Gemma 4 26B A4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.13,"output":0.4,"cache_read":0.05}},"openai-gpt-53-codex":{"id":"openai-gpt-53-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"qwen-3-8-2-4t-a95b":{"id":"qwen-3-8-2-4t-a95b","name":"Qwen 3.8 2.4T","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3.75,"output":18.75,"cache_read":0.375}},"openai-gpt-56-sol":{"id":"openai-gpt-56-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.33,"output":0.48,"cache_read":0.16}},"xiaomi-mimo-v2-5":{"id":"xiaomi-mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-06-11","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.5,"output":15,"cache_read":0.5,"cache_write":0.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5}}},"openai-gpt-56-terra-pro":{"id":"openai-gpt-56-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32000},"cost":{"input":2.27,"output":6.8,"cache_read":0.34,"tiers":[{"input":4.53,"output":13.6,"cache_read":0.68,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":0.68}}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-10","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"openai-gpt-52":{"id":"openai-gpt-52","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-13","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":272000,"output":65536},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"claude-opus-4-8-fast":{"id":"claude-opus-4-8-fast","name":"Claude Opus 4.8 Fast","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"openai-gpt-56-luna-pro":{"id":"openai-gpt-56-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"z-ai-glm-5-3-flash":{"id":"z-ai-glm-5-3-flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"grok-4-20":{"id":"grok-4-20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"google-gemma-4-31b-it":{"id":"google-gemma-4-31b-it","name":"Google Gemma 4 31B Instruct","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-03","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.12,"output":0.36,"cache_read":0.09}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.325,"output":3.25}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-18","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.25,"output":5.0625,"cache_read":0.2125}},"qwen-3-8-flash":{"id":"qwen-3-8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.49,"cache_read":0.014}},"qwen3-coder-480b-a35b-instruct-turbo":{"id":"qwen3-coder-480b-a35b-instruct-turbo","name":"Qwen 3 Coder 480B Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"openai-gpt-4o-2024-11-20":{"id":"openai-gpt-4o-2024-11-20","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":3.125,"output":12.5}},"minimax-m27":{"id":"minimax-m27","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.375,"output":1.5,"cache_read":0.06875}},"qwen3-next-80b":{"id":"qwen3-next-80b","name":"Qwen 3 Next 80b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.35,"output":1.9}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.3125,"output":0.9375,"cache_read":0.03125}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":64000},"cost":{"input":3.75,"output":18.75,"cache_read":0.375,"cache_write":4.69}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"NVIDIA Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.3}},"zai-org-glm-4.7":{"id":"zai-org-glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.55,"output":2.65,"cache_read":0.11}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-19","last_updated":"2026-06-11","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.7,"output":3.75,"cache_read":0.07}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen 3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.75}},"z-ai-glm-5v-turbo":{"id":"z-ai-glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":1.5,"output":5,"cache_read":0.3}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-10","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":200000},"cost":{"input":2.27,"output":6.8,"cache_read":0.57,"tiers":[{"input":4.53,"output":13.6,"cache_read":1.13,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":1.13}}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"qwen-3-7-plus":{"id":"qwen-3-7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":3.301,"cache_read":0.33}},"nvidia-nemotron-3-ultra-550b-a55b":{"id":"nvidia-nemotron-3-ultra-550b-a55b","name":"NVIDIA Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.625,"output":3.125,"cache_read":0.1875}},"z-ai-glm-5-3":{"id":"z-ai-glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.75,"output":5.5,"cache_read":0.325}},"grok-4-20-multi-agent":{"id":"grok-4-20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"openai-gpt-56-luna":{"id":"openai-gpt-56-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"llama-3.3-70b":{"id":"llama-3.3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"release_date":"2025-04-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.7,"output":2.8}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-07-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3 VL 235B","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"openai-gpt-4o-mini-2024-07-18":{"id":"openai-gpt-4o-mini-2024-07-18","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1875,"output":0.75,"cache_read":0.09375}},"gemini-3-8-flash":{"id":"gemini-3-8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"openai-gpt-56-sol-pro":{"id":"openai-gpt-56-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}}}},"gmicloud":{"id":"gmicloud","env":["GMICLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.gmi-serving.com/v1","name":"GMI Cloud","doc":"https://docs.gmicloud.ai/inference-engine/api-reference/llm-api-reference","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":384000},"cost":{"input":0.112,"output":0.224,"cache_read":0.022}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.392,"output":2.784,"cache_read":0.116}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.979,"output":3.08,"cache_read":0.182}},"zai-org/GLM-5-FP8":{"id":"zai-org/GLM-5-FP8","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.855,"output":3.6,"cache_read":0.144}}}},"io-net":{"id":"io-net","env":["IOINTELLIGENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.intelligence.io.solutions/api/v1","name":"IO.NET","doc":"https://io.net/docs/guides/intelligence/io-intelligence","models":{"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar":{"id":"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":106000,"output":4096},"cost":{"input":0.22,"output":0.95,"cache_read":0.11,"cache_write":0.44}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8.75,"cache_read":1,"cache_write":4}},"mistralai/Devstral-Small-2505":{"id":"mistralai/Devstral-Small-2505","name":"Devstral Small 2505","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"mistralai/Mistral-Large-Instruct-2411":{"id":"mistralai/Mistral-Large-Instruct-2411","name":"Mistral Large Instruct 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":6,"cache_read":1,"cache_write":4}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.04,"cache_read":0.01,"cache_write":0.04}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0.25,"cache_write":1}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-15","last_updated":"2024-11-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.4,"output":1.75,"cache_read":0.2,"cache_write":0.8}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen 2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen 3 Next 80B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.8,"cache_read":0.05,"cache_write":0.2}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen 3 235B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.11,"output":0.6,"cache_read":0.055,"cache_write":0.22}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":430000,"output":4096},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.3}},"meta-llama/Llama-3.2-90B-Vision-Instruct":{"id":"meta-llama/Llama-3.2-90B-Vision-Instruct","name":"Llama 3.2 90B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.35,"output":0.4,"cache_read":0.175,"cache_write":0.7}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.13,"output":0.38,"cache_read":0.065,"cache_write":0.26}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":4096},"cost":{"input":0.03,"output":0.14,"cache_read":0.015,"cache_write":0.06}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.04,"output":0.4,"cache_read":0.02,"cache_write":0.08}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.55,"output":2.25,"cache_read":0.275,"cache_write":1.1}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-09-05","last_updated":"2024-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.39,"output":1.9,"cache_read":0.195,"cache_write":0.78}}}},"llmgateway":{"id":"llmgateway","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"DevPass (LLM Gateway)","doc":"https://llmgateway.io/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.2}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.38,"output":1.98,"cache_read":0.19,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":3.125}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"minimax-m2.1-lightning":{"id":"minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"gemini-pro-latest":{"id":"gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025,"cache_write":0}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"codestral-2508":{"id":"codestral-2508","name":"Codestral","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":0.9}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"seed-1-8-251228":{"id":"seed-1-8-251228","name":"Seed 1.8 (251228)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.08,"output":0.35,"cache_read":0.05}},"llama-4-scout-17b-instruct":{"id":"llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":2048},"cost":{"input":0.18,"output":0.59}},"qwen35-397b-a17b":{"id":"qwen35-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking (2507)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.3,"output":3}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11,"cache_write":0}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"gpt-4o-mini-transcribe":{"id":"gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":1.25,"output":5}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.27,"output":1.1}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"glm-4.6v-flashx":{"id":"glm-4.6v-flashx","name":"GLM-4.6V FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"ling-3.0-flash":{"id":"ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"qwen3-235b-a22b-fp8":{"id":"qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":8192},"cost":{"input":0.2,"output":0.8}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.08,"output":0.32,"cache_read":0.017,"cache_write":0.375}},"custom":{"id":"custom","name":"Custom Model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16,"cache_write":0}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":1050000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":228700,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"grok-4":{"id":"grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"seed-1-6-flash-250715":{"id":"seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.36,"output":0.87,"reasoning":8.4}},"llama-3.2-11b-instruct":{"id":"llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.07,"output":0.33}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"minimax-m2.5-highspeed":{"id":"minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"llama-3.2-3b-instruct":{"id":"llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"glm-4.5-x":{"id":"glm-4.5-x","name":"GLM-4.5 X","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"grok-4-7":{"id":"grok-4-7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-4o-transcribe":{"id":"gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":2.5,"output":10}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8,"cache_read":0.04,"cache_write":0.25}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"qwen-coder-plus":{"id":"qwen-coder-plus","name":"Qwen Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0.07,"output":0.27}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"ernie-4.5-vl-424b-a47b":{"id":"ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":123000},"cost":{"input":0.42,"output":1.25}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"minimax-text-01":{"id":"minimax-text-01","name":"MiniMax Text 01","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"glm-4.5-airx":{"id":"glm-4.5-airx","name":"GLM-4.5 AirX","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"hy-mt2-plus":{"id":"hy-mt2-plus","name":"Hy-MT2 Plus","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"Hy","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.074,"output":0.295}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.1,"output":0.1}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38,"cache_read":0.6,"cache_write":3.75}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.07,"output":0.19,"cache_read":0.015}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"qwen3-vl-flash":{"id":"qwen3-vl-flash","name":"Qwen3 VL Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"qwen3-vl-30b-a3b-instruct":{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4,"cache_read":0.08,"cache_write":0.5}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"llama-4-maverick-17b-instruct":{"id":"llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":2048},"cost":{"input":0.27,"output":0.85}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"atria-dawn-preview":{"id":"atria-dawn-preview","name":"Atria Dawn Preview","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"fugu-max":{"id":"fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.57,"output":2.3,"cache_read":0.5}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":0.72,"output":2.3,"cache_read":0.144,"cache_write":0}},"glm-4-32b-0414-128k":{"id":"glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.1}},"seed-1-6-250615":{"id":"seed-1-6-250615","name":"Seed 1.6 (250615)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.931,"output":2.93,"cache_read":0.173,"cache_write":0}},"qwen3-vl-235b-a22b-thinking":{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.98,"output":3.95}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"llama-3.1-70b-instruct":{"id":"llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"status":"beta","cost":{"input":0.72,"output":0.72}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct (2507)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.09,"output":0.58}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.2,"output":0.8}},"grok-4-20-beta-0309-reasoning":{"id":"grok-4-20-beta-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4-20-beta-0309-non-reasoning":{"id":"grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.15}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"seed-1-6-250915":{"id":"seed-1-6-250915","name":"Seed 1.6 (250915)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"qwen-plus-latest":{"id":"qwen-plus-latest","name":"Qwen Plus Latest","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":8192},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"fugu-ultra-v2.0":{"id":"fugu-ultra-v2.0","name":"Fugu Ultra v2.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.135,"output":0.4}},"llama-3-70b-instruct":{"id":"llama-3-70b-instruct","name":"Llama 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01,"cache_write":0}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"auto":{"id":"auto","name":"Auto Route","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"infomaniak":{"id":"infomaniak","env":["INFOMANIAK_API_KEY","INFOMANIAK_PRODUCT_ID"],"npm":"@ai-sdk/openai-compatible","api":"https://api.infomaniak.com/2/ai/${INFOMANIAK_PRODUCT_ID}/openai/v1","name":"Infomaniak","doc":"https://www.infomaniak.com/en/hosting/ai-services/open-source-models","models":{"bge_multilingual_gemma2":{"id":"bge_multilingual_gemma2","name":"BGE Multilingual Gemma2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-25","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"input":8000,"output":3584},"cost":{"input":0.08,"output":0}},"mini_lm_l12_v2":{"id":"mini_lm_l12_v2","name":"All-MiniLM-L12-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128,"input":128,"output":384},"cost":{"input":0,"output":0}},"swiss-ai/Apertus-v1.5-70B":{"id":"swiss-ai/Apertus-v1.5-70B","name":"Apertus v1.5 70B","description":"Open, ethically-sourced Swiss AI model for multilingual, multimodal chat and instruction following","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-08-01","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":8192},"status":"beta","cost":{"input":0.87,"output":3.1}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.25,"output":0.93}},"mistralai/Ministral-3-14B-Instruct-2512":{"id":"mistralai/Ministral-3-14B-Instruct-2512","name":"Ministral 3 14B Instruct","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":25600},"status":"beta","cost":{"input":0.37,"output":0.5}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8","name":"Nemotron 3 Nano 30B A3B FP8","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":262144},"status":"beta","cost":{"input":0.06,"output":0.25}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":32768},"cost":{"input":0.25,"output":0.5}},"Qwen/Qwen3.5-122B-A10B-FP8":{"id":"Qwen/Qwen3.5-122B-A10B-FP8","name":"Qwen3.5 122B-A10B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"cost":{"input":0.5,"output":3.97}},"Qwen/Qwen3.5-397B-A17B-FP8":{"id":"Qwen/Qwen3.5-397B-A17B-FP8","name":"Qwen3.5 397B-A17B FP8","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"status":"beta","cost":{"input":0.99,"output":4.46}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"status":"beta","cost":{"input":0.74,"output":3.72}}}},"inception":{"id":"inception","env":["INCEPTION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptionlabs.ai/v1/","name":"Inception","doc":"https://docs.inceptionlabs.ai/get-started/models","models":{"mercury-2.5":{"id":"mercury-2.5","name":"Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11-01","release_date":"2026-09-08","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"mercury-edit-2":{"id":"mercury-edit-2","name":"Mercury Edit 2","description":"Code editing dLLM for autocomplete (FIM) and next-edit suggestions","family":"mercury","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}}}},"lilac":{"id":"lilac","env":["LILAC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.getlilac.com/v1","name":"Lilac","doc":"https://docs.getlilac.com/inference/models","models":{"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":262100},"cost":{"input":0.11,"output":0.35}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":524288},"cost":{"input":0.9,"output":3,"cache_read":0.27}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.28,"output":1.1,"cache_read":0.05}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.2}}}},"fastrouter":{"id":"fastrouter","env":["FASTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://go.fastrouter.ai/api/v1","name":"FastRouter","doc":"https://fastrouter.ai/models","models":{"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1.2}},"deepseek-ai/deepseek-r1-distill-llama-70b":{"id":"deepseek-ai/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.14}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/veo3.1-fast":{"id":"google/veo3.1-fast","name":"Veo 3.1 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/veo3.1":{"id":"google/veo3.1","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"google/veo3.1-lite":{"id":"google/veo3.1-lite","name":"Veo 3.1 Lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"google/imagen-4.0-ultra":{"id":"google/imagen-4.0-ultra","name":"Imagen 4 Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4.0-fast":{"id":"google/imagen-4.0-fast","name":"Imagen 4 Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.0375}},"bytedance/seedance-2":{"id":"bytedance/seedance-2","name":"Seedance 2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":4096,"output":0}},"wanx/wan-v2-6":{"id":"wanx/wan-v2-6","name":"Wan 2.6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":true,"limit":{"context":400000,"output":0}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48}},"leonardo-ai/lucid-realism":{"id":"leonardo-ai/lucid-realism","name":"Lucid Realism","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"leonardo-ai/lucid-origin":{"id":"leonardo-ai/lucid-origin","name":"Lucid Origin","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"x-ai/grok-4":{"id":"x-ai/grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.75,"cache_write":15}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT Realtime 1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32000,"output":4096},"cost":{"input":4,"output":16}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.05,"output":0.2}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.55,"output":2.2}},"sarvam/sarvam-105b":{"id":"sarvam/sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"sarvam/sarvam-30b":{"id":"sarvam/sarvam-30b","name":"Sarvam 30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.1}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.95,"output":3.15}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.05,"output":3.5}}}},"cloudflare-ai-gateway":{"id":"cloudflare-ai-gateway","env":["CLOUDFLARE_API_TOKEN","CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_GATEWAY_ID"],"npm":"ai-gateway-provider","name":"Cloudflare AI Gateway","doc":"https://developers.cloudflare.com/ai-gateway/","models":{"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"alibaba/qwen3.5-397b-a17b":{"id":"alibaba/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":5,"cache_read":0.625}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}}}},"github-copilot":{"id":"github-copilot","env":["GITHUB_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.githubcopilot.com","name":"GitHub Copilot","doc":"https://docs.github.com/en/copilot","models":{"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":64000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"claude-opus-4.7":{"id":"claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":224000,"output":32000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"claude-sonnet-4.6":{"id":"claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":32000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"mai-code-1-flash-picker":{"id":"mai-code-1-flash-picker","name":"MAI-Code-1-Flash","description":"Microsoft coding model built for fast, efficient assistance in everyday developer workflows","family":"mai","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-06-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":136000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":24000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":128000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"mai-code-1.1-flash":{"id":"mai-code-1.1-flash","name":"MAI-Code-1.1-Flash","description":"Microsoft coding model with native vision support, optimized for fast and efficient software development","family":"mai","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":264000,"input":128000,"output":64000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"zhipuai":{"id":"zhipuai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/paas/v4","name":"Zhipu AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.3-flashx":{"id":"glm-5.3-flashx","name":"GLM-5.3-FlashX","description":"High-speed GLM-5.3-Flash serving option for coding and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":5,"output":22,"cache_read":1.2,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v-flash":{"id":"glm-4.6v-flash","name":"GLM-4.6V-Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}}}},"jalapeno":{"id":"jalapeno","env":["JALAPENO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jalapeno-cloud.ai/v1","name":"Jalapeno Cloud","doc":"https://www.jalapeno-cloud.ai/docs/","models":{"Qwen3.5-27B":{"id":"Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.38,"output":4.4}},"Qwen3.5-122B-A10B":{"id":"Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.3,"output":1.5}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":180224},"cost":{"input":0.6,"output":3}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":271360,"output":262144},"cost":{"input":0.95,"output":4}},"Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.15,"output":1.5}},"Qwen3.5-397B-A17B":{"id":"Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"Qwen3.5-35B-A3B":{"id":"Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"Hy3":{"id":"Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.6,"output":3.38}}}},"perplexity-agent":{"id":"perplexity-agent","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.perplexity.ai/v1","name":"Perplexity Agent","doc":"https://docs.perplexity.ai/docs/agent-api/models","models":{"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32000},"cost":{"input":0.25,"output":2.5}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"moonshot-ai/kimi-k2.7-code":{"id":"moonshot-ai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot-ai/kimi-k3":{"id":"moonshot-ai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"xai/grok-4-1-fast-non-reasoning":{"id":"xai/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.25,"output":2.5,"cache_read":0.0625}}}},"fireworks-ai":{"id":"fireworks-ai","env":["FIREWORKS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.fireworks.ai/inference/v1/","name":"Fireworks AI","doc":"https://fireworks.ai/docs/","models":{"accounts/fireworks/routers/kimi-latest":{"id":"accounts/fireworks/routers/kimi-latest","name":"Kimi Latest","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3.75,"output":18.75,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/routers/qwen-max-latest":{"id":"accounts/fireworks/routers/qwen-max-latest","name":"Qwen Max Latest (Qwen3.8 Max)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3,"output":9,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/routers/kimi-k3-fast":{"id":"accounts/fireworks/routers/kimi-k3-fast","name":"Kimi K3 Fast","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"accounts/fireworks/routers/glm-flash-latest":{"id":"accounts/fireworks/routers/glm-flash-latest","name":"GLM Flash Latest (GLM 5.3 Flash)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":0.1875,"output":0.625,"cache_read":0.0375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"accounts/fireworks/routers/minimax-latest":{"id":"accounts/fireworks/routers/minimax-latest","name":"MiniMax Latest","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"experimental":{"modes":{"priority":{"cost":{"input":0.45,"output":1.8,"cache_read":0.09},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"accounts/fireworks/routers/glm-fast-latest":{"id":"accounts/fireworks/routers/glm-fast-latest","name":"GLM 5.3 Fast (Latest)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048572,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.39}},"accounts/fireworks/routers/deepseek-pro-latest":{"id":"accounts/fireworks/routers/deepseek-pro-latest","name":"DeepSeek Pro Latest","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":1.65,"output":4.95,"cache_read":0.055},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"accounts/fireworks/routers/glm-5p3-fast":{"id":"accounts/fireworks/routers/glm-5p3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048572,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.39}},"accounts/fireworks/routers/glm-latest":{"id":"accounts/fireworks/routers/glm-latest","name":"GLM Latest","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":262144},"experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.325},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"accounts/fireworks/routers/kimi-fast-latest":{"id":"accounts/fireworks/routers/kimi-fast-latest","name":"Kimi Fast Latest","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"accounts/fireworks/routers/glm-5p2-fast":{"id":"accounts/fireworks/routers/glm-5p2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"accounts/fireworks/routers/deepseek-flash-latest":{"id":"accounts/fireworks/routers/deepseek-flash-latest","name":"DeepSeek Flash Latest","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/qwen3p7-plus":{"id":"accounts/fireworks/models/qwen3p7-plus","name":"Qwen 3.7 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08}},"accounts/fireworks/models/deepseek-v4-flash-vision-exp":{"id":"accounts/fireworks/models/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.65,"output":4.95,"cache_read":0.055},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/minimax-m3":{"id":"accounts/fireworks/models/minimax-m3","name":"MiniMax-M3","description":"Fireworks text-only MiniMax coding model for long-context reasoning and agent tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"experimental":{"modes":{"priority":{"cost":{"input":0.45,"output":1.8,"cache_read":0.09},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"accounts/fireworks/models/deepseek-v4p1-flash":{"id":"accounts/fireworks/models/deepseek-v4p1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/kimi-k2p6":{"id":"accounts/fireworks/models/kimi-k2p6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.5,"output":6,"cache_read":0.22},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"accounts/fireworks/models/nemotron-3-ultra-nvfp4":{"id":"accounts/fireworks/models/nemotron-3-ultra-nvfp4","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"accounts/fireworks/models/kimi-k3":{"id":"accounts/fireworks/models/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3.75,"output":18.75,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/models/glm-5p3":{"id":"accounts/fireworks/models/glm-5p3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":262144},"experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.325},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"accounts/fireworks/models/kimi-k2p7-code":{"id":"accounts/fireworks/models/kimi-k2p7-code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.425,"output":6,"cache_read":0.285},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"accounts/fireworks/models/glm-5p3-flash":{"id":"accounts/fireworks/models/glm-5p3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":0.1875,"output":0.625,"cache_read":0.0375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"accounts/fireworks/models/minimax-m2p7":{"id":"accounts/fireworks/models/minimax-m2p7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","provider":{"body":{"service_tier":"priority"}},"cost":{"input":1.2,"output":1.2,"cache_read":0.6}},"accounts/fireworks/models/qwen3p8-2p4t-a95b":{"id":"accounts/fireworks/models/qwen3p8-2p4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/models/muse-glimmer-30b":{"id":"accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"accounts/fireworks/models/inkling":{"id":"accounts/fireworks/models/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"accounts/fireworks/models/deepseek-v4-pro":{"id":"accounts/fireworks/models/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.2,"output":1.2,"cache_read":0.6},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.2,"output":1.2,"cache_read":0.6}},"accounts/fireworks/models/gpt-oss-120b":{"id":"accounts/fireworks/models/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"experimental":{"modes":{"priority":{"cost":{"input":0.18,"output":0.72,"cache_read":0.018},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"accounts/fireworks/models/glm-5p2":{"id":"accounts/fireworks/models/glm-5p2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.175},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"accounts/fireworks/models/qwen3p8-max":{"id":"accounts/fireworks/models/qwen3p8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3,"output":9,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b":{"id":"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}}}},"opper":{"id":"opper","env":["OPPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.opper.ai/v3/compat","name":"Opper","doc":"https://opper.ai/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.4,"output":2}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.5811,"output":3}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.69732,"output":2.78928}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.62708,"output":5.811}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":524288}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.25,"output":0.66}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.11622,"output":0.488124}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.5811,"output":2.3244}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.07}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.46488,"output":2.44062}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125,"tiers":[{"input":6.6,"output":24.75,"cache_read":0.66,"cache_write":8.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6.6,"output":24.75,"cache_read":0.66,"cache_write":8.25}}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5811,"output":2.44062}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":1.78812,"output":3.57624}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":1.1622,"output":4.88124}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.75,"output":4.6488,"cache_read":0.44}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"stackit":{"id":"stackit","env":["STACKIT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1","name":"STACKIT","doc":"https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models","models":{"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-05-17","last_updated":"2025-05-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":37000,"output":4096},"cost":{"input":0.53,"output":0.76}},"Qwen/Qwen3-VL-Embedding-8B":{"id":"Qwen/Qwen3-VL-Embedding-8B","name":"Qwen3-VL Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.09,"output":0.09}},"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8","name":"Qwen3-VL 235B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":218000,"output":16384},"cost":{"input":1.76,"output":2.05}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.53,"output":0.76}},"intfloat/e5-mistral-7b-instruct":{"id":"intfloat/e5-mistral-7b-instruct","name":"E5 Mistral 7B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.02,"output":0.02}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.29}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":8192},"cost":{"input":0.53,"output":0.76}},"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic":{"id":"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.53,"output":0.76}}}},"crof":{"id":"crof","env":["CROF_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://crof.ai/v1","name":"CrofAI","doc":"https://crof.ai/docs","models":{"greg-2-super":{"id":"greg-2-super","name":"Greg 2 Super","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":1.5,"output":5,"cache_read":0.25}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.2,"cache_read":0.007}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro (0813)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.01}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash (New)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.1,"cache_read":0.003}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.04,"output":0.15,"cache_read":0.008}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.03}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.5,"output":1.99,"cache_read":0.05}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.3,"output":1.05,"cache_read":0.05}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.12,"output":0.21,"cache_read":0.003}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.55,"output":2.25,"cache_read":0.05}},"greg-1-mini":{"id":"greg-1-mini","name":"Greg 1 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.07,"output":0.15,"cache_read":0.01}},"greg-2-ultra":{"id":"greg-2-ultra","name":"Greg 2 Ultra","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":3,"output":10,"cache_read":0.5}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":1.75,"cache_read":0.07}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.04}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":8,"cache_read":0.25}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.18,"output":0.35,"cache_read":0.04}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.07,"output":0.22,"cache_read":0.01}},"greg-rp":{"id":"greg-rp","name":"Greg (Roleplay)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.45,"output":2.15,"cache_read":0.08,"cache_write":0}},"kimi-k3-eco":{"id":"kimi-k3-eco","name":"Kimi K3 Eco","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":1,"output":4,"cache_read":0.1}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.003}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":1.4,"cache_read":0.06}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":0.8,"cache_read":0.003,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}}}},"crusoe":{"id":"crusoe","env":["CRUSOE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.crusoecloud.com/v1","name":"Crusoe","doc":"https://docs.crusoecloud.com/managed-inference/overview","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.3,"output":1.83,"cache_read":0.3,"input_audio":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4,"cache_read":0.14}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.8,"cache_read":0.11}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.25,"output":0.75,"cache_read":0.13}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2,"cache_read":0.05}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.35}},"zai/GLM-5.1":{"id":"zai/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4.4,"cache_read":0.25}},"zai/GLM-5.2":{"id":"zai/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"empiriolabs":{"id":"empiriolabs","env":["EMPIRIOLABS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.empiriolabs.ai/v1","name":"EmpirioLabs AI","doc":"https://docs.empiriolabs.ai","models":{"glm-5-1":{"id":"glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.165,"tiers":[{"input":1.1,"output":3.851,"cache_read":0.22,"tier":{"type":"context","size":32000}}]}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13,"cache_read":0.045}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"kimi-k2-7-code-highspeed":{"id":"kimi-k2-7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.9,"output":8,"cache_read":1.9}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.424,"output":1.272,"cache_read":0.424}},"mistral-small-4":{"id":"mistral-small-4","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"gemma-3-27b":{"id":"gemma-3-27b","name":"Gemma 3 27B","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"qwen3-8-max-0902":{"id":"qwen3-8-max-0902","name":"Qwen3.8 Max 0902","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"seed-2-0-pro":{"id":"seed-2-0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.63,"output":3.79,"cache_read":0.63,"tiers":[{"input":1.26,"output":7.58,"cache_read":1.26,"tier":{"type":"context","size":128000}}]}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":256000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.4,"tiers":[{"input":1.2,"output":4.8,"cache_read":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":1.2}}},"qwen3-6-flash":{"id":"qwen3-6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.25,"tiers":[{"input":1,"output":4,"cache_read":1,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1,"output":4,"cache_read":1}}},"qwen3-5-27b":{"id":"qwen3-5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.086,"output":0.688,"cache_read":0.086,"tiers":[{"input":0.258,"output":2.064,"cache_read":0.258,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"glm-4-6v-flash":{"id":"glm-4-6v-flash","name":"GLM 4.6V Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0}},"seed-2-0-mini":{"id":"seed-2-0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.12,"output":0.5,"cache_read":0.12,"tiers":[{"input":0.24,"output":1,"cache_read":0.24,"tier":{"type":"context","size":128000}}]}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":524288},"cost":{"input":0.225,"output":0.9,"cache_read":0.045,"tiers":[{"input":0.45,"output":1.8,"cache_read":0.09,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.45,"output":1.8,"cache_read":0.09}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"qwen3-5-4b":{"id":"qwen3-5-4b","name":"Qwen3.5 4B","description":"Qwen3.5 4B is a low-cost multimodal reasoning model with 256K context, image and video input, function tools, and structured output.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-02","last_updated":"2026-03-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.04,"output":0.07,"cache_read":0.02}},"glm-5-3":{"id":"glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"fugu-ultra-v1-1":{"id":"fugu-ultra-v1-1","name":"Fugu Ultra v1.1","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"step-3-5-flash-2603":{"id":"step-3-5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":1.032,"cache_read":0.172,"tiers":[{"input":0.43,"output":2.58,"cache_read":0.43,"tier":{"type":"context","size":128000}}]}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.63,"output":3.13,"cache_read":0.63}},"deepseek-v3-2":{"id":"deepseek-v3-2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.57,"output":1.71,"cache_read":0.57}},"glm-4-7-flash":{"id":"glm-4-7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16000},"cost":{"input":0.8939,"output":3.7131,"cache_read":0.1788}},"gemma-4-26b-a4b":{"id":"gemma-4-26b-a4b","name":"Gemma 4 26B-A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.29,"cache_read":0.025}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.07,"output":0.42,"cache_read":0.035}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.057,"output":0.459,"cache_read":0.057,"tiers":[{"input":0.229,"output":1.835,"cache_read":0.229,"tier":{"type":"context","size":128000}}]}},"muse-spark-1-2":{"id":"muse-spark-1-2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"qwen3-6-plus":{"id":"qwen3-6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.5,"tiers":[{"input":2,"output":6,"cache_read":2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":2}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":3}},"qwen3-5-122b-a10b":{"id":"qwen3-5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.115,"output":0.917,"cache_read":0.115,"tiers":[{"input":0.287,"output":2.294,"cache_read":0.287,"tier":{"type":"context","size":128000}}]}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.08,"output":5.52,"cache_read":1.08,"tiers":[{"input":2.16,"output":11.04,"cache_read":2.16,"tier":{"type":"context","size":32000}},{"input":2.7,"output":13.8,"cache_read":2.7,"tier":{"type":"context","size":128000}}]}},"glm-4-5-flash":{"id":"glm-4-5-flash","name":"GLM 4.5 Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":98304},"cost":{"input":0,"output":0}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.075}},"step-3-5-flash":{"id":"step-3-5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":131072},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"qwen3-8-omni-flash":{"id":"qwen3-8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":0.94,"cache_read":0.3}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.412564,"output":2.475384,"cache_read":0.412564}},"qwen3-8-27b":{"id":"qwen3-8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.17,"output":0.5,"cache_read":0.08}},"muse-spark-1-1":{"id":"muse-spark-1-1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"glm-5-2":{"id":"glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"seed-2-0-code":{"id":"seed-2-0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.4,"tiers":[{"input":0.8,"output":4.8,"cache_read":0.8,"tier":{"type":"context","size":128000}}]}},"qwen3-7-max":{"id":"qwen3-7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":2.5}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.7,"output":1.4,"cache_read":0.014}},"qwen3-5-flash":{"id":"qwen3-5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.09,"output":0.368,"cache_read":0.09}},"seed-2-0-lite":{"id":"seed-2-0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.31,"output":2.5,"cache_read":0.31,"tiers":[{"input":0.62,"output":5,"cache_read":0.62,"tier":{"type":"context","size":128000}}]}},"muse-spark-1-3":{"id":"muse-spark-1-3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.175,"output":4.35,"cache_read":0.018}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3,"cache_read":1.65}},"fugu-ultra-v1-0":{"id":"fugu-ultra-v1-0","name":"Fugu Ultra v1.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":7.5,"output":45,"cache_read":1.5,"tiers":[{"input":15,"output":67.5,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":15,"output":67.5,"cache_read":3}}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.03}},"qwen3-5-plus":{"id":"qwen3-5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.36,"output":2.21,"cache_read":0.36,"tiers":[{"input":1.08,"output":6.62,"cache_read":1.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.08,"output":6.62,"cache_read":1.08}}},"qwen3-7-flash":{"id":"qwen3-7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"tier":{"type":"context","size":256000}}]}},"qwen3-8-flash":{"id":"qwen3-8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.16}},"qwen3-6-max-preview":{"id":"qwen3-6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88,"cache_read":1.31,"tiers":[{"input":1.97,"output":11.82,"cache_read":1.97,"tier":{"type":"context","size":128000}}]}},"fugu-ultra-v2-0":{"id":"fugu-ultra-v2-0","name":"Fugu Ultra v2.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"klokintegration":{"id":"klokintegration","env":["KLOKINTEGRATION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-gw.klok.ipaas.se/proxy/kloker-key/v1","name":"klokintegration.se","doc":"https://klokintegration.se/docs/ai-api","models":{"Kloker-Integration-Developer":{"id":"Kloker-Integration-Developer","name":"Kloker Integration Developer","description":"Knows the customer integration environment and Klok best practices. Opinionated about implementation. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection. Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker-Integration-Architect":{"id":"Kloker-Integration-Architect","name":"Kloker Integration Architect","description":"Knows the customer integration environment and Klok best practices. Opinionated about structure. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection (data contracts, CloudEvents, event-driven flows). Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker":{"id":"Kloker","name":"Kloker","description":"Cheap general model with a clean context. Nothing from the customer environment is packed in. It tracks the current best open source model. The Klok team verifies it and upgrades it periodically.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}}}},"privatemode-ai":{"id":"privatemode-ai","env":["PRIVATEMODE_API_KEY","PRIVATEMODE_ENDPOINT"],"npm":"@ai-sdk/openai-compatible","api":"http://localhost:8080/v1","name":"Privatemode AI","doc":"https://docs.privatemode.ai/api/overview","models":{"kimi-latest":{"id":"kimi-latest","name":"Kimi (latest)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"status":"deprecated","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"status":"deprecated","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"voxtral-mini-3b":{"id":"voxtral-mini-3b","name":"Voxtral Mini 3B","description":"Speech-to-text model for audio transcription, translation, and audio understanding","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07","last_updated":"2025-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.00462,"output":0}},"qwen3-embedding-4b":{"id":"qwen3-embedding-4b","name":"Qwen3-Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-06","last_updated":"2025-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2560},"cost":{"input":0.1502,"output":0}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper large-v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.01618,"output":0}},"glm-flash-latest":{"id":"glm-flash-latest","name":"GLM Flash (latest)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":0.8897,"output":4.4718,"cache_read":0.0924}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":0.8897,"output":4.4718,"cache_read":0.0924}},"glm-latest":{"id":"glm-latest","name":"GLM (latest)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.4969,"output":1.9644,"cache_read":0.0462}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"beta","cost":{"input":0.8897,"output":1.4675,"cache_read":0.0924}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}}}},"minimax-coding-plan":{"id":"minimax-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax Token Plan (minimax.io)","doc":"https://platform.minimax.io/docs/token-plan/intro","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"kimi-code-plan-global":{"id":"kimi-code-plan-global","env":["KIMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kimi.ai/coding/v1","name":"Kimi For Coding (kimi.ai)","doc":"https://www.kimi.ai/code/docs/en/kimi-code/models.html","models":{"kimi-for-coding-highspeed":{"id":"kimi-for-coding-highspeed","name":"Kimi For Coding HighSpeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3-256k":{"id":"k3-256k","name":"Kimi K3-256K","description":"256K-context version of Kimi K3, reducing token consumption for shorter coding sessions","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-for-coding":{"id":"kimi-for-coding","name":"kimi-for-coding","description":"Kimi coding model with more efficient thinking and up to 1M context, available through Kimi Code","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3":{"id":"k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"inferx":{"id":"inferx","env":["INFERX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://model.inferx.net/endpoints/v1","name":"InferX","doc":"https://model.inferx.net/endpoints","models":{"gemma-4-31B-it-fp8":{"id":"gemma-4-31B-it-fp8","name":"Gemma 4 31B IT FP8","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8-no-thinking":{"id":"Qwen3-Coder-Next-FP8-no-thinking","name":"Qwen3-Coder-Next-FP8-no-thinking","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}},"Devstral-2-123B-Instruct-2512-int4-AutoRound":{"id":"Devstral-2-123B-Instruct-2512-int4-AutoRound","name":"Devstral-2-123B-Instruct-2512-int4-AutoRound","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"Agents-A1":{"id":"Agents-A1","name":"Agents-A1","description":"35B MoE agentic model built for long-horizon search, engineering, and scientific reasoning tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"Ornith-1.0-35B-FP8":{"id":"Ornith-1.0-35B-FP8","name":"Ornith-1.0-35B-FP8","description":"Large coding-reasoning model for agentic software tasks and RL search","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-25","last_updated":"2026-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-FP8":{"id":"Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Qwen3.6-27B-FP8":{"id":"Qwen3.6-27B-FP8","name":"Qwen3.6 27B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-fp8-no-thinking":{"id":"Qwen3.6-35B-A3B-fp8-no-thinking","name":"Qwen3.6-35B-A3B-fp8-no-thinking","description":"Qwen3.6-35B-A3B-fp8 disable thinking","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8":{"id":"Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Embedding-8B":{"id":"Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":0},"cost":{"input":0,"output":0}},"mimo-v25":{"id":"mimo-v25","name":"mimo-v25","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}}}},"umans-ai-coding-plan":{"id":"umans-ai-coding-plan","env":["UMANS_AI_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI Coding Plan","doc":"https://app.umans.ai/offers/code/docs","models":{"umans-qwen3.6-35b-a3b":{"id":"umans-qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-glm-5.3-flash":{"id":"umans-glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"databricks":{"id":"databricks","env":["DATABRICKS_HOST","DATABRICKS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1","name":"Databricks","doc":"https://docs.databricks.com/aws/en/machine-learning/foundation-models/","models":{"databricks-claude-opus-4-5":{"id":"databricks-claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-sonnet-4-6":{"id":"databricks-claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gemini-3-pro":{"id":"databricks-gemini-3-pro","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-kimi-k2-7-code":{"id":"databricks-kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"databricks-gpt-5-6-luna":{"id":"databricks-gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"databricks-claude-opus-4-1":{"id":"databricks-claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"databricks-gpt-5-mini":{"id":"databricks-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"databricks-gemini-2-5-flash":{"id":"databricks-gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"databricks-claude-haiku-4-5":{"id":"databricks-claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"databricks-claude-sonnet-4-5":{"id":"databricks-claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-4":{"id":"databricks-gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-gpt-5-6-sol":{"id":"databricks-gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"databricks-glm-5-2":{"id":"databricks-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"databricks-gpt-5-4-nano":{"id":"databricks-gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"databricks-gpt-5-5":{"id":"databricks-gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"databricks-gemini-3-1-flash-lite":{"id":"databricks-gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"databricks-gemini-3-flash":{"id":"databricks-gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"databricks-claude-opus-4-7":{"id":"databricks-claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-opus-4-6":{"id":"databricks-claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-sonnet-4":{"id":"databricks-claude-sonnet-4","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-1":{"id":"databricks-gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-oss-20b":{"id":"databricks-gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2}},"databricks-gpt-5-4-mini":{"id":"databricks-gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"databricks-gemini-3-1-pro":{"id":"databricks-gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-gpt-5-6-terra":{"id":"databricks-gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-gemini-2-5-pro":{"id":"databricks-gemini-2-5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"databricks-gpt-5":{"id":"databricks-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-5-nano":{"id":"databricks-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"databricks-gpt-oss-120b":{"id":"databricks-gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.072,"output":0.28}},"databricks-gpt-5-2":{"id":"databricks-gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}}}},"modal":{"id":"modal","env":["MODAL_PROXY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.us-west.modal.direct/v1","name":"Modal","doc":"https://modal.com/docs/guide/endpoints","models":{"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.45,"output":1.5,"cache_read":0.09}},"thinkingmachines/Inkling-NVFP4":{"id":"thinkingmachines/Inkling-NVFP4","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.2,"output":5,"cache_read":0.27}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8-Max","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1010000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"reasoning":15,"cache_read":0.3}}}},"lucidquery":{"id":"lucidquery","env":["LUCIDQUERY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lucidquery.com/v1","name":"LucidQuery","doc":"https://lucidquery.com/docs","models":{"lucidquery-nexus-coder":{"id":"lucidquery-nexus-coder","name":"LucidQuery Nexus Coder","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"lucid","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-01","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":250000,"output":60000},"cost":{"input":2,"output":5}},"lucidquery-agi-01-frontier":{"id":"lucidquery-agi-01-frontier","name":"AGI-01 Frontier","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":4.5,"output":22}},"lucidquery-agi-01-swift":{"id":"lucidquery-agi-01-swift","name":"AGI-01 Swift","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":2.5,"output":15}},"lucidnova-rf1-100b":{"id":"lucidnova-rf1-100b","name":"LucidNova RF1 100B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-09-16","release_date":"2024-12-28","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":120000,"output":8000},"cost":{"input":2,"output":5}}}},"atomic-chat":{"id":"atomic-chat","env":["ATOMIC_CHAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1337/v1","name":"Atomic Chat","doc":"https://atomic.chat","models":{"Meta-Llama-3_1-8B-Instruct-GGUF":{"id":"Meta-Llama-3_1-8B-Instruct-GGUF","name":"Meta Llama 3.1 8B Instruct (GGUF)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0,"output":0}},"Qwen3_5-9B-Q4_K_M":{"id":"Qwen3_5-9B-Q4_K_M","name":"Qwen 3.5 9B (Q4_K_M)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"Qwen3_5-9B-MLX-4bit":{"id":"Qwen3_5-9B-MLX-4bit","name":"Qwen 3.5 9B (MLX 4-bit)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma-4-E4B-it-MLX-4bit":{"id":"gemma-4-E4B-it-MLX-4bit","name":"Gemma 4 E4B Instruct (MLX 4-bit)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma-4-E4B-it-IQ4_XS":{"id":"gemma-4-E4B-it-IQ4_XS","name":"Gemma 4 E4B Instruct (IQ4_XS)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}}}},"umans-ai":{"id":"umans-ai","env":["UMANS_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI","doc":"https://app.umans.ai/offers/code/docs/orgs","models":{"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"umans-glm-5.3-flash":{"id":"umans-glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}}}},"sakana":{"id":"sakana","env":["SAKANA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sakana.ai/v1","name":"Sakana AI","doc":"https://console.sakana.ai/models","models":{"fugu":{"id":"fugu","name":"Fugu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"fugu-ultra-20260615":{"id":"fugu-ultra-20260615","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana-namazu":{"id":"sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}}}},"deepinfra":{"id":"deepinfra","env":["DEEPINFRA_API_KEY"],"npm":"@ai-sdk/deepinfra","name":"Deep Infra","doc":"https://deepinfra.com/models","models":{"ByteDance/Seed-2.0-mini":{"id":"ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02,"tiers":[{"input":0.2,"output":0.8,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-code":{"id":"ByteDance/Seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-pro":{"id":"ByteDance/Seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.09,"output":0.18,"cache_read":0.018}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":0.8}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.4,"output":0.4}},"nvidia/Nemotron-3-Nano-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.02,"output":0.1}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.75,"output":2.4,"cache_read":0.14}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.6,"output":2.08,"cache_read":0.12}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.5,"output":2,"cache_read":0.1}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"tiers":[{"input":5,"output":15,"cache_read":1,"tier":{"type":"context","size":32000}},{"input":6.25,"output":18.5,"cache_read":1.25,"tier":{"type":"context","size":128000}}]}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":2.5,"cache_read":0.05}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.6}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"Qwen/Qwen3.8-Max":{"id":"Qwen/Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":1.65,"output":4.951,"cache_read":0.206}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.4}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.2}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.55}},"Qwen/Qwen3.8-Flash":{"id":"Qwen/Qwen3.8-Flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.113,"output":0.382,"cache_read":0.0141}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":1.1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen 3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.45,"output":3,"cache_read":0.22}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.1,"output":0.95}},"Qwen/Qwen3-Max":{"id":"Qwen/Qwen3-Max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32000}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128000}}]}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.15,"output":1.15,"cache_read":0.03}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.28,"output":1.1,"cache_read":0.056}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.25,"output":1,"cache_read":0.05}},"meta-llama/Llama-4-Scout-17B-16E-Instruct":{"id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Llama 4 Scout 17B","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.2,"output":0.8}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.03,"output":0.14}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.037,"output":0.17}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.68,"output":3.4,"cache_read":0.136}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.75,"output":3.5,"cache_read":0.15}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.85,"output":14.25,"cache_read":0.285}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.13,"output":0.53,"cache_read":0.033}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}}}},"wafer.ai":{"id":"wafer.ai","env":["WAFER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://pass.wafer.ai/v1","name":"Wafer","doc":"https://docs.wafer.ai/wafer-pass","models":{"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"General Language Model 5.1 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.1,"cache_write":0}},"glm5.2-fast":{"id":"glm5.2-fast","name":"GLM5.2-Fast","description":"The same model served for high TPS.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":10.25,"cache_read":0.5,"cache_write":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.1,"cache_read":0.2,"cache_write":0}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.14,"output":4.8,"cache_read":0.19,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.07,"cache_write":0,"tiers":[{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0}}}}},"kilo":{"id":"kilo","env":["KILO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kilo.ai/api/gateway","name":"Kilo Gateway","doc":"https://kilo.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.475,"output":4.425,"cache_read":0.295,"cache_write":1.84375}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.0975,"output":0.78}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen: Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.2275,"output":0.91}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.425,"output":2.55,"cache_read":0.085,"cache_write":0.53125}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1625,"output":1.3}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.975,"output":4.875}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.2925,"output":1.4625}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.1495,"output":0.598}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.39,"output":2.34}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.7}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.13,"output":0.52}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen3.8-27b:free":{"id":"qwen/qwen3.8-27b:free","name":"Qwen: Qwen3.8 27B (free)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":0.13,"output":0.52}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.13,"output":0.52}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen: Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.26,"output":1.04}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B (retires Oct 8)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Anthropic: Claude Fable Latest ($$$$)","description":"This model always redirects to the latest model in the Claude Fable family.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Anthropic: Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Anthropic: Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Anthropic: Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph: Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph: Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek: DeepSeek V4 Flash Latest","description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.04,"output":0.2,"cache_read":0.01}},"~deepseek/deepseek-pro-latest":{"id":"~deepseek/deepseek-pro-latest","name":"DeepSeek: DeepSeek Pro Latest","description":"This model always redirects to the latest model in the DeepSeek Pro family.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":384000},"cost":{"input":0.5544,"output":1.6632,"cache_read":0.01848}},"~deepseek/deepseek-flash-latest":{"id":"~deepseek/deepseek-flash-latest","name":"DeepSeek: DeepSeek Flash Latest","description":"This model always redirects to the latest model in the DeepSeek Flash family.","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.12,"output":0.48,"cache_read":0.0036}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots Studio: Dots3-Note Preview (free)","description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"xAI: Grok Latest","description":"This model always redirects to the latest Grok model from xAI.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"PrismML: Ternary Bonsai 2 27B","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.5}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Poolside: Laguna XS 2.1 (free)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Poolside: Laguna S 2.1 (free)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"kwaipilot/kat-coder-pro-v2":{"id":"kwaipilot/kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":144000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash:free":{"id":"stepfun/step-3.7-flash:free","name":"StepFun: Step 3.7 Flash (free)","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters per token. The model supports a 256K context window and exposes selectable reasoning levels (high/medium/low), letting callers trade off speed, cost, and depth of reasoning. Designed for coding, agentic workflows, structured outputs, and long-context productivity tasks.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Mistral: Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral: Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral: Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.004,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax: MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.4,"output":2.2}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax: MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"output":900172},"cost":{"input":0.2,"output":1.1}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"NVIDIA: Nemotron 3.5 Lightning (free)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.065,"output":0.18}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"NVIDIA: Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.08,"output":0.45}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"NVIDIA: Nemotron 3 Ultra (free)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"NVIDIA: Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"NVIDIA: Nemotron 3.5 Content Safety (free)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":182520},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.042,"output":0.22}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.375,"output":1.875,"reasoning":1.875,"cache_read":0.0375,"cache_write":0.020833}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":4.5,"reasoning":4.5,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.16}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.15,"output":1.25,"reasoning":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.041667}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows.","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace: Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace: Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex AGI: Nex-N2.5-Mini (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex AGI: Nex-N2.5-Pro (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Thinking Machines: Inkling Small (free)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":471859},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3686},"cost":{"input":0.08,"output":0.11}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Meta: Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Meta: Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.3,"output":1.1,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron: Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed: Seed 2.1 Turbo","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"ByteDance Seed: Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Inception: Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.2,"output":0.75,"cache_read":0.02}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Inception: Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Writer: Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Google: Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Google: Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Microsoft: Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Sakana: Fugu Max","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Sakana: Fugu Ultra v2","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"MoonshotAI: Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.7,"output":8.5,"cache_read":0.17}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"IBM: Granite 4.2 8B","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the cost-efficient tier of the V4.1 family. DeepSeek reports that it exceeds V4 Pro on performance, speed, and task...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.32,"output":0.89}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek: R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B (retires Sep 28)","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp (retires Sep 28)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus (retires Sep 28)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":147456},"cost":{"input":0.25,"output":1}},"~openai/gpt-terra-latest":{"id":"~openai/gpt-terra-latest","name":"OpenAI: GPT Terra Latest","description":"This model always redirects to the latest model in the OpenAI GPT Terra family.","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"~openai/gpt-sol-latest":{"id":"~openai/gpt-sol-latest","name":"OpenAI: GPT Sol Latest","description":"This model always redirects to the latest model in the OpenAI GPT Sol family.","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"~openai/gpt-luna-latest":{"id":"~openai/gpt-luna-latest","name":"OpenAI: GPT Luna Latest","description":"This model always redirects to the latest model in the OpenAI GPT Luna family.","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"~openai/gpt-astra-latest":{"id":"~openai/gpt-astra-latest","name":"OpenAI: GPT Astra Latest ($$$$)","description":"This model always redirects to the latest model in the OpenAI GPT Astra family.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"OpenAI: GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon: Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Amazon: Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon: Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Amazon: Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon: Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"inclusionAI: Ling 3.0 Flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"inclusionAI: Ling 3.0 Flash Fin","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"inclusionAI: Ling 3.0 Flash Fin (free)","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"inclusionAI: Ling 3.0 Flash Sante (free)","description":"Ling 3.0 Flash Sante is a health and medicine-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl:free":{"id":"inclusionai/ling-3.0-flash-vl:free","name":"inclusionAI: Ling 3.0 Flash VL (free)","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"inclusionAI: Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"mancer/weaver":{"id":"mancer/weaver","name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"openrouter/free":{"id":"openrouter/free","name":"OpenRouter Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":0,"output":0}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"status":"beta","cost":{"input":0,"output":0}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["audio","image","pdf","text","video"],"output":["image","text"]},"open_weights":false,"limit":{"context":2000000,"output":32768},"cost":{"input":0,"output":0}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"kilo-auto/free":{"id":"kilo-auto/free","name":"Auto Free","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0,"cache_write":0}},"kilo-auto/efficient":{"id":"kilo-auto/efficient","name":"Auto Efficient","description":"Routes each request to the cheapest model that gets the job done, based on continuously benchmarked accuracy and cost.","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"kilo-auto/small":{"id":"kilo-auto/small","name":"Auto Small","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"reasoning":0,"cache_read":0.005}},"kilo-auto/frontier":{"id":"kilo-auto/frontier","name":"Auto Frontier","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"reasoning":0,"cache_read":0.5,"cache_write":6.25}},"kilo-auto/balanced":{"id":"kilo-auto/balanced","name":"Auto Balanced","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"SpaceXAI: Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"Grok 4.7 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.02,"output":0.04}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Meta: Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":16384},"cost":{"input":0.1875,"output":0.6525}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Meta: Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Nous: Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nousresearch","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI: o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI: o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"OpenAI: GPT-6 Astra Pro ($$$$)","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"OpenAI: GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.018,"output":0.09}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"OpenAI: GPT-5 Image ($$$$)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"OpenAI: GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"OpenAI: GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"Z.ai: GLM Flash Latest","description":"This model always redirects to the latest model in the GLM Flash family.","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":102400},"cost":{"input":0.075,"output":0.25,"cache_read":0.02}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"Z.ai: GLM Latest","description":"This model always redirects to the latest GLM model from Z.ai.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.7728,"output":2.4288,"cache_read":0.14352}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"MoonshotAI: Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Inference.net: Schematron V2 Small","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Inference.net: Schematron V2 Turbo","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"Cohere: North Mini Code (free)","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a":{"id":"cohere/command-a","name":"Cohere: Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Upstage: Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Upstage: Solar Pro 4","description":"Solar Pro 4 is a large language model from Upstage. It is suited for agentic workflows, office productivity, document-intensive work, and coding.","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Tencent: Hy-MT2-30B-A3B","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Tencent: Hy-MT2-7B","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Tencent: Hy-MT2-1.8B","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LiquidAI: LFM2.5-2.6B (free)","description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"Z.ai: GLM 5.3 FlashX","description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":102400},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5.2:free":{"id":"z-ai/glm-5.2:free","name":"Z.ai: GLM 5.2 (free)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0,"output":0}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.0605,"output":0.4}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Perplexity: Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Perplexity: Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Perplexity: Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Perplexity: Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}},"stealth/claude-opus-4.8":{"id":"stealth/claude-opus-4.8","name":"Stealth: Claude Opus 4.8 (20% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Claude Opus 4.8 is offered at 20% lower cost than standard Claude Opus 4.8 pricing and is not served by Anthropic or Kilo Code.","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/qwen3.6-plus":{"id":"stealth/qwen3.6-plus","name":"Stealth: Qwen3.6 Plus (50% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Qwen3.6 Plus is offered at 50% lower cost than standard Qwen3.6 Plus pricing and is not served by Alibaba or Kilo Code. Note: a surcharge applies to long-context workloads exceeding 256K input tokens.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":0,"cache_read":0.025,"cache_write":0.3125}},"stealth/claude-opus-4.7":{"id":"stealth/claude-opus-4.7","name":"Stealth: Claude Opus 4.7 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/claude-sonnet-4.6":{"id":"stealth/claude-sonnet-4.6","name":"Stealth: Claude Sonnet 4.6 (20% off)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.4,"output":12,"reasoning":0,"cache_read":0.24,"cache_write":3}},"stealth/claude-opus-4.6":{"id":"stealth/claude-opus-4.6","name":"Stealth: Claude Opus 4.6 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}}}},"alibaba-coding-plan":{"id":"alibaba-coding-plan","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-intl.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/coding-plan","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"submodel":{"id":"submodel","env":["SUBMODEL_INSTAGEN_ACCESS_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.submodel.ai/v1","name":"submodel","doc":"https://submodel.gitbook.io","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.5,"output":2.15}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5}},"zai-org/GLM-4.5-FP8":{"id":"zai-org/GLM-4.5-FP8","name":"GLM 4.5 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.3}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.6}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}}}},"openreason":{"id":"openreason","env":["OPENREASON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openreason.app/v1","name":"OpenReason","doc":"https://openreason.app/docs","models":{"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1371,"output":0.2743}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1055,"output":0.422}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.0022,"output":4.22}}}},"azure":{"id":"azure","env":["AZURE_RESOURCE_NAME","AZURE_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":15,"output":120}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek-V4-Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.19,"output":0.51}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"GPT-Image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-image-1":{"id":"gpt-image-1","name":"GPT-Image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-image-2.5-sunburst":{"id":"gpt-image-2.5-sunburst","name":"GPT Image 2.5 Sunburst","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"status":"beta"},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek-V4-Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":1.74,"output":3.48}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"gpt-image-2.5-flare":{"id":"gpt-image-2.5-flare","name":"GPT Image 2.5 Flare","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"amazon-bedrock":{"id":"amazon-bedrock","env":["AWS_ACCESS_KEY_ID","AWS_SECRET_ACCESS_KEY","AWS_REGION","AWS_BEARER_TOKEN_BEDROCK"],"npm":"@ai-sdk/amazon-bedrock","name":"Amazon Bedrock","doc":"https://docs.aws.amazon.com/bedrock/latest/userguide/models-supported.html","models":{"moonshotai.kimi-k2.5":{"id":"moonshotai.kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262143,"output":16384},"cost":{"input":0.6,"output":3}},"global.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"global.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (Global)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"us.anthropic.claude-opus-5":{"id":"us.anthropic.claude-opus-5","name":"Claude Opus 5 (US)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"eu.amazon.nova-pro-v1:0":{"id":"eu.amazon.nova-pro-v1:0","name":"Nova Pro (EU)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.92,"output":3.68,"cache_read":0.23,"cache_write":0.92}},"us.writer.palmyra-x4-v1:0":{"id":"us.writer.palmyra-x4-v1:0","name":"Palmyra X4 (US)","description":"Enterprise language model for workflow automation, coding, data analysis, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-10-09","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"us.anthropic.claude-opus-4-6-v1":{"id":"us.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (US)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"google.gemma-4-31b":{"id":"google.gemma-4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.14,"output":0.4}},"us.xai.grok-4.6":{"id":"us.xai.grok-4.6","name":"Grok 4.6 (US)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"eu.mistral.pixtral-large-2502-v1:0":{"id":"eu.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (EU)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"qwen.qwen3-coder-next":{"id":"qwen.qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.5,"output":1.2}},"global.openai.gpt-5.6-luna":{"id":"global.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (Global)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"global.anthropic.claude-opus-4-6-v1":{"id":"global.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (Global)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai.gpt-5.5":{"id":"openai.gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":5.5,"output":33,"cache_read":0.55}},"us-gov.openai.gpt-oss-20b-1:0":{"id":"us-gov.openai.gpt-oss-20b-1:0","name":"gpt-oss-20b (GovCloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.084,"output":0.36}},"qwen.qwen3-coder-30b-a3b-v1:0":{"id":"qwen.qwen3-coder-30b-a3b-v1:0","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}},"global.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"global.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (Global)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen.qwen3-235b-a22b-2507-v1:0":{"id":"qwen.qwen3-235b-a22b-2507-v1:0","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.22,"output":0.88}},"mistral.ministral-3-3b-instruct":{"id":"mistral.ministral-3-3b-instruct","name":"Ministral 3 3B","description":"Compact open vision-language model for edge deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1,"output":0.1}},"us-gov.openai.gpt-oss-120b-1:0":{"id":"us-gov.openai.gpt-oss-120b-1:0","name":"gpt-oss-120b (GovCloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.18,"output":0.72}},"global.anthropic.claude-sonnet-4-6":{"id":"global.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Global)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"openai.gpt-5.4":{"id":"openai.gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"mistral.pixtral-large-2502-v1:0":{"id":"mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"mistral.mistral-large-3-675b-instruct":{"id":"mistral.mistral-large-3-675b-instruct","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5,"output":1.5}},"anthropic.claude-opus-4-5-20251101-v1:0":{"id":"anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.amazon.nova-micro-v1:0":{"id":"us.amazon.nova-micro-v1:0","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"jp.anthropic.claude-opus-4-7":{"id":"jp.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (JP)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"eu.anthropic.claude-sonnet-5":{"id":"eu.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"apac.amazon.nova-micro-v1:0":{"id":"apac.amazon.nova-micro-v1:0","name":"Nova Micro (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.037,"output":0.148,"cache_read":0.00925,"cache_write":0.037}},"nvidia.nemotron-nano-9b-v2":{"id":"nvidia.nemotron-nano-9b-v2","name":"NVIDIA Nemotron Nano 9B v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.06,"output":0.23}},"au.anthropic.claude-sonnet-4-6":{"id":"au.anthropic.claude-sonnet-4-6","name":"AU Anthropic Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"anthropic.claude-opus-4-7":{"id":"anthropic.claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral.ministral-3-8b-instruct":{"id":"mistral.ministral-3-8b-instruct","name":"Ministral 3 8B","description":"Compact open vision-language model for edge deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.15}},"au.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"au.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (AU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"us.openai.gpt-5.6-sol":{"id":"us.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (US)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"eu.amazon.nova-lite-v1:0":{"id":"eu.amazon.nova-lite-v1:0","name":"Nova Lite (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.069,"output":0.276,"cache_read":0.01725,"cache_write":0.069}},"anthropic.claude-opus-5":{"id":"anthropic.claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-opus-4-6-v1":{"id":"eu.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"apac.amazon.nova-pro-v1:0":{"id":"apac.amazon.nova-pro-v1:0","name":"Nova Pro (APAC)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.84,"output":3.36,"cache_read":0.21,"cache_write":0.84}},"anthropic.claude-sonnet-4-6":{"id":"anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"apac.amazon.nova-lite-v1:0":{"id":"apac.amazon.nova-lite-v1:0","name":"Nova Lite (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.063,"output":0.252,"cache_read":0.01575,"cache_write":0.063}},"mistral.voxtral-mini-3b-2507":{"id":"mistral.voxtral-mini-3b-2507","name":"Voxtral Mini 3B 2507","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":4096},"cost":{"input":0.04,"output":0.04}},"google.gemma-4-26b-a4b":{"id":"google.gemma-4-26b-a4b","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.13,"output":0.4}},"nvidia.nemotron-nano-12b-v2":{"id":"nvidia.nemotron-nano-12b-v2","name":"NVIDIA Nemotron Nano 12B v2 VL BF16","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.6}},"nvidia.nemotron-nano-3-30b":{"id":"nvidia.nemotron-nano-3-30b","name":"NVIDIA Nemotron Nano 3 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.06,"output":0.24}},"eu.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"eu.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"minimax.minimax-m2.1":{"id":"minimax.minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta.llama3-3-70b-instruct-v1:0":{"id":"meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"deepseek.v3-v1:0":{"id":"deepseek.v3-v1:0","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"eu.anthropic.claude-opus-5":{"id":"eu.anthropic.claude-opus-5","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"anthropic.claude-sonnet-5":{"id":"anthropic.claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"us.writer.palmyra-x5-v1:0":{"id":"us.writer.palmyra-x5-v1:0","name":"Palmyra X5 (US)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"input":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"google.gemma-4-e2b":{"id":"google.gemma-4-e2b","name":"Gemma 4 E2B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.04,"output":0.08}},"us.meta.llama4-maverick-17b-instruct-v1:0":{"id":"us.meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct (US)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.24,"output":0.97}},"meta.llama3-1-8b-instruct-v1:0":{"id":"meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"minimax.minimax-m2":{"id":"minimax.minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204608,"output":128000},"cost":{"input":0.3,"output":1.2}},"global.anthropic.claude-opus-5":{"id":"global.anthropic.claude-opus-5","name":"Claude Opus 5 (Global)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"eu.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen.qwen3-32b-v1:0":{"id":"qwen.qwen3-32b-v1:0","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.15,"output":0.6}},"writer.palmyra-x4-v1:0":{"id":"writer.palmyra-x4-v1:0","name":"Palmyra X4","description":"Enterprise language model for workflow automation, coding, data analysis, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-10-09","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"us.amazon.nova-pro-v1:0":{"id":"us.amazon.nova-pro-v1:0","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"google.gemma-3-12b-it":{"id":"google.gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.09,"output":0.29}},"au.anthropic.claude-opus-4-8":{"id":"au.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (AU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"jp.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (JP)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"eu.amazon.nova-2-lite-v1:0":{"id":"eu.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (EU)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.374,"output":3.157,"cache_read":0.0935,"cache_write":0.374}},"eu.anthropic.claude-opus-4-8":{"id":"eu.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.anthropic.claude-opus-5":{"id":"jp.anthropic.claude-opus-5","name":"Claude Opus 5 (JP)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"mistral.ministral-3-14b-instruct":{"id":"mistral.ministral-3-14b-instruct","name":"Ministral 14B 3.0","description":"Open vision-language model for efficient local deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"openai.gpt-oss-safeguard-20b":{"id":"openai.gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.2}},"global.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"global.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (Global)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"global.amazon.nova-2-lite-v1:0":{"id":"global.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (Global)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.3}},"eu.amazon.nova-micro-v1:0":{"id":"eu.amazon.nova-micro-v1:0","name":"Nova Micro (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.04,"output":0.16,"cache_read":0.01,"cache_write":0.04}},"openai.gpt-5.6-luna":{"id":"openai.gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"anthropic.claude-opus-4-6-v1":{"id":"anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"openai.gpt-oss-20b-1:0":{"id":"openai.gpt-oss-20b-1:0","name":"gpt-oss-20b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"us.amazon.nova-premier-v1:0":{"id":"us.amazon.nova-premier-v1:0","name":"Nova Premier (US)","description":"Multimodal model for complex analysis, long-context understanding, tool use, and model distillation","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":10000},"status":"deprecated","cost":{"input":2.5,"output":12.5,"cache_read":0.625,"cache_write":2.5}},"qwen.qwen3-vl-235b-a22b":{"id":"qwen.qwen3-vl-235b-a22b","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262000},"cost":{"input":0.53,"output":2.66}},"amazon.nova-2-lite-v1:0":{"id":"amazon.nova-2-lite-v1:0","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.0825,"cache_write":0.33}},"global.xai.grok-4.6":{"id":"global.xai.grok-4.6","name":"Grok 4.6 (Global)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"global.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"global.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (Global)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"amazon.nova-lite-v1:0":{"id":"amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"anthropic.claude-opus-4-8":{"id":"anthropic.claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.amazon.nova-2-lite-v1:0":{"id":"us.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (US)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.0825,"cache_write":0.33}},"us.openai.gpt-5.6-terra":{"id":"us.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (US)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"us.meta.llama3-3-70b-instruct-v1:0":{"id":"us.meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct (US)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"us.meta.llama3-1-70b-instruct-v1:0":{"id":"us.meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct (US)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"amazon.nova-pro-v1:0":{"id":"amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"us.anthropic.claude-opus-4-7":{"id":"us.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (US)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"au.anthropic.claude-opus-4-6-v1":{"id":"au.anthropic.claude-opus-4-6-v1","name":"AU Anthropic Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"writer.palmyra-x5-v1:0":{"id":"writer.palmyra-x5-v1:0","name":"Palmyra X5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"input":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"global.openai.gpt-5.6-sol":{"id":"global.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (Global)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai.gpt-5.6-sol":{"id":"openai.gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"global.anthropic.claude-opus-4-8":{"id":"global.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (Global)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax.minimax-m2.5":{"id":"minimax.minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":98304},"cost":{"input":0.3,"output":1.2}},"openai.gpt-oss-120b-1:0":{"id":"openai.gpt-oss-120b-1:0","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"eu.anthropic.claude-opus-4-7":{"id":"eu.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (EU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.meta.llama4-scout-17b-instruct-v1:0":{"id":"us.meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct (US)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":10000000,"output":8192},"cost":{"input":0.17,"output":0.66}},"us.openai.gpt-5.6-luna":{"id":"us.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (US)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"us.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"us.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (US)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"moonshot.kimi-k2-thinking":{"id":"moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262143,"output":16000},"cost":{"input":0.6,"output":2.5}},"anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"deepseek.r1-v1:0":{"id":"deepseek.r1-v1:0","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"mistral.magistral-small-2509":{"id":"mistral.magistral-small-2509","name":"Magistral Small 1.2","description":"Open multimodal reasoning model for transparent analysis of text and images","family":"magistral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":40000},"cost":{"input":0.5,"output":1.5}},"us.anthropic.claude-fable-5":{"id":"us.anthropic.claude-fable-5","name":"Claude Fable 5 (US)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"eu.anthropic.claude-fable-5":{"id":"eu.anthropic.claude-fable-5","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"us.openai.gpt-6-astra":{"id":"us.openai.gpt-6-astra","name":"GPT-6 Astra (US)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"us.anthropic.claude-fable-5-1":{"id":"us.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (US)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"meta.llama4-scout-17b-instruct-v1:0":{"id":"meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":10000000,"output":8192},"cost":{"input":0.17,"output":0.66}},"jp.amazon.nova-2-lite-v1:0":{"id":"jp.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (JP)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.396,"output":3.311,"cache_read":0.099,"cache_write":0.396}},"google.gemma-3-27b-it":{"id":"google.gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":8192},"cost":{"input":0.23,"output":0.38}},"amazon.nova-micro-v1:0":{"id":"amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"us.mistral.pixtral-large-2502-v1:0":{"id":"us.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"anthropic.claude-fable-5":{"id":"anthropic.claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"xai.grok-4.6":{"id":"xai.grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"global.anthropic.claude-fable-5":{"id":"global.anthropic.claude-fable-5","name":"Claude Fable 5 (Global)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"au.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"au.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (AU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"eu.anthropic.claude-sonnet-4-6":{"id":"eu.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"in.openai.gpt-5.6-terra":{"id":"in.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (India)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"jp.anthropic.claude-opus-4-8":{"id":"jp.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (JP)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"eu.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"eu.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (EU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"qwen.qwen3-next-80b-a3b":{"id":"qwen.qwen3-next-80b-a3b","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262000},"cost":{"input":0.15,"output":1.2}},"us.anthropic.claude-sonnet-5":{"id":"us.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (US)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"us.amazon.nova-lite-v1:0":{"id":"us.amazon.nova-lite-v1:0","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"global.anthropic.claude-opus-4-7":{"id":"global.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (Global)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"qwen.qwen3-coder-480b-a35b-v1:0":{"id":"qwen.qwen3-coder-480b-a35b-v1:0","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.45,"output":1.8}},"openai.gpt-5.6-terra":{"id":"openai.gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"nvidia.nemotron-super-3-120b":{"id":"nvidia.nemotron-super-3-120b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.65}},"zai.glm-4.7-flash":{"id":"zai.glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"google.gemma-3-4b-it":{"id":"google.gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.04,"output":0.08}},"global.openai.gpt-5.6-terra":{"id":"global.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (Global)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"zai.glm-5":{"id":"zai.glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2}},"openai.gpt-oss-safeguard-120b":{"id":"openai.gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"mistral.devstral-2-123b":{"id":"mistral.devstral-2-123b","name":"Devstral 2 123B","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.4,"output":2}},"openai.gpt-6-astra":{"id":"openai.gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"us.anthropic.claude-opus-4-1-20250805-v1:0":{"id":"us.anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"us.anthropic.claude-sonnet-4-6":{"id":"us.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (US)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"mistral.voxtral-small-24b-2507":{"id":"mistral.voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.1,"output":0.3}},"openai.gpt-oss-20b":{"id":"openai.gpt-oss-20b","name":"gpt-oss-20b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.07,"output":0.3}},"meta.llama4-maverick-17b-instruct-v1:0":{"id":"meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.24,"output":0.97}},"zai.glm-4.7":{"id":"zai.glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"ca.amazon.nova-lite-v1:0":{"id":"ca.amazon.nova-lite-v1:0","name":"Nova Lite (CA)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.064,"output":0.256,"cache_read":0.016,"cache_write":0.064}},"us.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"us.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"au.anthropic.claude-opus-4-7":{"id":"au.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (AU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.anthropic.claude-sonnet-4-6":{"id":"jp.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (JP)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"us.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"us.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (US)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"us.deepseek.r1-v1:0":{"id":"us.deepseek.r1-v1:0","name":"DeepSeek-R1 (US)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"us.anthropic.claude-opus-4-8":{"id":"us.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (US)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"au.anthropic.claude-opus-5":{"id":"au.anthropic.claude-opus-5","name":"Claude Opus 5 (AU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"anthropic.claude-opus-4-1-20250805-v1:0":{"id":"anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"apac.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"apac.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (APAC)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"jp.anthropic.claude-sonnet-5":{"id":"jp.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (JP)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"au.anthropic.claude-sonnet-5":{"id":"au.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (AU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"xai.grok-4.3":{"id":"xai.grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-06-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"openai.gpt-oss-120b":{"id":"openai.gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.15,"output":0.6}},"meta.llama3-1-70b-instruct-v1:0":{"id":"meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"global.anthropic.claude-fable-5-1":{"id":"global.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (Global)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"us.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"us.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (US)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"anthropic.claude-fable-5-1":{"id":"anthropic.claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"global.anthropic.claude-sonnet-5":{"id":"global.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (Global)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"us.meta.llama3-1-8b-instruct-v1:0":{"id":"us.meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct (US)","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"jp.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"jp.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (JP)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"global.openai.gpt-6-astra":{"id":"global.openai.gpt-6-astra","name":"GPT-6 Astra (Global)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"in.openai.gpt-5.6-luna":{"id":"in.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (India)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"eu.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"eu.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"deepseek.v3.2":{"id":"deepseek.v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.62,"output":1.85}}}},"merge-gateway":{"id":"merge-gateway","env":["MERGE_GATEWAY_API_KEY"],"npm":"merge-gateway-ai-sdk-provider","api":"https://api-gateway.merge.dev/v1/ai-sdk","name":"Merge Gateway","doc":"https://docs.merge.dev/merge-gateway","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.825,"output":2.4755,"cache_read":0.165}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.574,"output":2.294,"cache_read":0.1148}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.276,"output":1.651,"cache_read":0.0552}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.086,"output":0.688,"cache_read":0.0172}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.057,"output":0.459,"cache_read":0.020357}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.022,"output":0.216,"cache_read":0.0044}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.029,"output":0.287,"cache_read":0.0058}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.143,"output":1.434,"cache_read":0.0286}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.8,"cache_read":0.075}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.172,"output":1.032,"cache_read":0.0344}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.289,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485,"cache_read":0.0496}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434,"cache_read":0.0718}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":1010000},"cost":{"input":2.5,"output":6.25,"cache_read":0.5}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.287,"cache_read":0.023}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.115,"output":0.917,"cache_read":0.023}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.165,"output":0.99,"cache_read":0.033}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":1.076,"cache_read":0.0216}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.0574}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3-VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":2.867,"cache_read":0.0574}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3-VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.15785}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.6,"cache_read":0.05}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":1.8}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.688,"cache_read":0.023}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nemotron Nano 9B","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.06,"output":0.23}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0,"output":0}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3-7-sonnet-20250219":{"id":"anthropic/claude-3-7-sonnet-20250219","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-1-20250805":{"id":"anthropic/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-20250514":{"id":"anthropic/claude-opus-4-20250514","name":"Claude Opus 4 (20250514)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-20250514":{"id":"anthropic/claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (20251101)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.13,"output":0.4}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.08}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":2,"output":12}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-2.5-computer-use-preview-10-2025":{"id":"google/gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview (10-2025)","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":1.25,"output":10}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.08,"output":0.45,"cache_read":0.04}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":4096},"cost":{"input":0.15,"output":0}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Gemini 3.1 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B It","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.14,"output":0.4}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.09,"output":0.29}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":32000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.22,"output":0.22}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.99,"output":0.99}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":0.5,"cache_read":0.11}},"bytedance/dola-seed-2.0-code-preview":{"id":"bytedance/dola-seed-2.0-code-preview","name":"Dola Seed 2.0 Code (preview)","description":"Preview coding model for repository understanding, refactors, and engineering tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"bytedance/dola-seed-2.0-code":{"id":"bytedance/dola-seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4}},"bytedance/dola-seed-2.0-lite":{"id":"bytedance/dola-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Efficient Seed model for general chat, analysis, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-28","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":2}},"bytedance/dola-seed-2.0-pro":{"id":"bytedance/dola-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Higher-capability Seed model for complex chat, analysis, and production tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"bytedance/dola-seed-2.0-mini":{"id":"bytedance/dola-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Low-cost Seed model for general chat, extraction, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.4}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"Enterprise multimodal model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.6,"output":6}},"writer/palmyra-x4":{"id":"writer/palmyra-x4","name":"Palmyra X4","description":"Enterprise language model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-10-09","last_updated":"2024-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":2.5,"output":10}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":2.9,"output":14,"cache_read":0.3}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"deepseek/deepseek-v4-flash-0731-fast":{"id":"deepseek/deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"deepseek/deepseek-v4-flash-0423":{"id":"deepseek/deepseek-v4-flash-0423","name":"DeepSeek V4 Flash 0423","description":"Initial DeepSeek V4 Flash snapshot for economical reasoning, coding, and million-token agent workloads","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.139,"output":0.278}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.003625}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.035,"output":0.07,"cache_read":0.007}},"deepseek/deepseek-v3":{"id":"deepseek/deepseek-v3","name":"DeepSeek V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.035,"output":0.07,"cache_read":0.007}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":1.35,"output":5.4}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":0.28,"output":0.45,"cache_read":0.14}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":41000},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.2,"cache_read":0.02}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.07,"output":0.2,"cache_read":0,"cache_write":0}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.15,"output":0.6}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.36}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.3-chat-latest":{"id":"openai/gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.6,"output":2.5}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+ 08-2024","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A 03-2025","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B 12-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R 08-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.7":{"id":"xai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":50000}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.05,"output":3.3,"cache_read":0.195}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.015,"output":0.05,"cache_read":0.003}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"Glm 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11,"cache_write":0}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.7,"output":2.2,"cache_read":0.13}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"mistral/devstral-small-2507":{"id":"mistral/devstral-small-2507","name":"Devstral Small","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/devstral-medium-2507":{"id":"mistral/devstral-medium-2507","name":"Devstral Medium","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-large-2411":{"id":"mistral/mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/pixtral-large-latest":{"id":"mistral/pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}}}},"deepseek":{"id":"deepseek","env":["DEEPSEEK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.deepseek.com","name":"DeepSeek","doc":"https://api-docs.deepseek.com/quick_start/pricing","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"reasoning":0.87,"cache_read":0.003625}},"deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}}}},"kimi-code-plan-cn":{"id":"kimi-code-plan-cn","env":["KIMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kimi.com/coding/v1","name":"Kimi For Coding (kimi.com)","doc":"https://www.kimi.com/code/docs/en/kimi-code/models.html","models":{"kimi-for-coding-highspeed":{"id":"kimi-for-coding-highspeed","name":"Kimi For Coding HighSpeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3-256k":{"id":"k3-256k","name":"Kimi K3-256K","description":"256K-context version of Kimi K3, reducing token consumption for shorter coding sessions","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-for-coding":{"id":"kimi-for-coding","name":"kimi-for-coding","description":"Kimi coding model with more efficient thinking and up to 1M context, available through Kimi Code","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3":{"id":"k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"abacus":{"id":"abacus","env":["ABACUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://routellm.abacus.ai/v1","name":"Abacus","doc":"https://abacus.ai/help/api","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"qwen-2.5-coder-32b":{"id":"qwen-2.5-coder-32b","name":"Qwen 2.5 Coder 32B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.79,"output":0.79}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.2,"output":1.5}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"claude-3-7-sonnet-20250219":{"id":"claude-3-7-sonnet-20250219","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"kimi-k2-turbo-preview":{"id":"kimi-k2-turbo-preview","name":"Kimi K2 Turbo Preview","description":"Fast Kimi model for responsive chat, coding help, and agent loops","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":0.15,"output":8}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1.2,"output":6}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"grok-4-0709":{"id":"grok-4-0709","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":3,"output":15}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.3-codex-xhigh":{"id":"gpt-5.3-codex-xhigh","name":"GPT-5.3 Codex XHigh","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.5}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3}},"route-llm":{"id":"route-llm","name":"RouteLLM","description":"RouteLLM routes prompts to an appropriate Abacus-backed text-generation model","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.59,"output":0.79}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"Grok 4 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":40}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-15","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":0.4}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.74,"output":3.48,"cache_read":0.15}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":96000},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.29,"output":1.2}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":0.38}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.09,"output":0.29}},"Qwen/QwQ-32B":{"id":"Qwen/QwQ-32B","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.4,"output":0.4}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.32,"output":3.2}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.55,"output":1.66}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta-llama/Meta-Llama-3.1-8B-Instruct":{"id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.05}},"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"id":"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo","name":"Llama 3.1 405B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":3.5,"output":3.5}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.14,"output":0.59}},"meta-llama/Meta-Llama-3.3-70B-Instruct":{"id":"meta-llama/Meta-Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.59,"output":0.79}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.08,"output":0.44}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"blueclaw":{"id":"blueclaw","env":["BLUECLAW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.blueclaw.network/v1","name":"Blue Claw","doc":"https://blueclaw.network","models":{"Qwen3.6-27B":{"id":"Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"status":"beta"},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta"}}},"kosmik":{"id":"kosmik","env":["KOSMIK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.koscompute.com/v1","name":"Kosmik Compute","doc":"https://api.koscompute.com/docs/","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.35,"output":2.2,"cache_read":0.09}}}},"opencode":{"id":"opencode","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/v1","name":"OpenCode Zen","doc":"https://opencode.ai/docs/zen","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"nemotron-3-ultra-free":{"id":"nemotron-3-ultra-free","name":"Nemotron 3 Ultra Free","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"hy3-preview-free":{"id":"hy3-preview-free","name":"Hy3 preview Free","description":"Legacy model retained for compatibility with older integrations","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"grok-code":{"id":"grok-code","name":"Grok Code Fast 1","description":"Legacy model retained for compatibility with older integrations","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-20","last_updated":"2025-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.3-contributor-free":{"id":"muse-spark-1.3-contributor-free","name":"Muse Spark 1.3 Free","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"north-mini-code-free":{"id":"north-mini-code-free","name":"North Mini Code Free","description":"Cohere coding model for practical software engineering and agentic edits","family":"north-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.1}},"minimax-m2.1-free":{"id":"minimax-m2.1-free","name":"MiniMax-M2.1 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"mimo-v2.6-flash-free":{"id":"mimo-v2.6-flash-free","name":"MiMo-V2.6-Flash Free","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"longcat-2.0-free":{"id":"longcat-2.0-free","name":"LongCat-2.0 Free","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash-free":{"id":"deepseek-v4-flash-free","name":"DeepSeek V4 Flash Free","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"laguna-s-2.1-free":{"id":"laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Legacy model retained for compatibility with older integrations","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"minimax-m3-free":{"id":"minimax-m3-free","name":"MiniMax-M3 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1,"output":2,"cache_read":0.2}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.45,"output":1.8}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"muse-spark-1.2-contributor-free":{"id":"muse-spark-1.2-contributor-free","name":"Muse Spark 1.2 Free","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"x-preview-f-free":{"id":"x-preview-f-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"ling-2.6-flash-free":{"id":"ling-2.6-flash-free","name":"Ling 2.6 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"ling-flash-free","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":32800},"status":"deprecated","cost":{"input":0,"output":0}},"gemini-3-pro":{"id":"gemini-3-pro","name":"Gemini 3 Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"nemotron-3.5-lightning-free":{"id":"nemotron-3.5-lightning-free","name":"Nemotron 3.5 Lightning Free","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"hy3-free":{"id":"hy3-free","name":"Hy3 Free","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":190000,"input":192000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"kimi-k2.5-free":{"id":"kimi-k2.5-free","name":"Kimi K2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"kimi-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"ring-2.6-1t-free":{"id":"ring-2.6-1t-free","name":"Ring 2.6 1T Free","description":"Legacy model retained for compatibility with older integrations","family":"ring-1t-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":66000},"status":"deprecated","cost":{"input":0,"output":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"mimo-v2.5-free":{"id":"mimo-v2.5-free","name":"MiMo V2.5 Free","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"claude-3-5-haiku":{"id":"claude-3-5-haiku","name":"Claude Haiku 3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"nemotron-3-super-free":{"id":"nemotron-3-super-free","name":"Nemotron 3 Super Free","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"big-pickle":{"id":"big-pickle","name":"Big Pickle","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"big-pickle","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":160000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"ling-3.0-flash-fin-free":{"id":"ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin Free","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3,"cache_read":0.08}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"ling-3.0-flash-free":{"id":"ling-3.0-flash-free","name":"Ling-3.0-flash Free","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"trinity-large-preview-free":{"id":"trinity-large-preview-free","name":"Trinity Large Preview","description":"Legacy model retained for compatibility with older integrations","family":"trinity","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-27","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0,"output":0}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.84,"cache_read":0.145}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"glm-4.7-free":{"id":"glm-4.7-free","name":"GLM-4.7 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5-free":{"id":"glm-5-free","name":"GLM-5 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"mimo-v2-flash-free":{"id":"mimo-v2-flash-free","name":"MiMo V2 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-flash-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"minimax-m2.5-free":{"id":"minimax-m2.5-free","name":"MiniMax-M2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-omni-free":{"id":"mimo-v2-omni-free","name":"MiMo V2 Omni Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-omni-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro-free":{"id":"mimo-v2-pro-free","name":"MiMo V2 Pro Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-pro-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"qwen3.6-plus-free":{"id":"qwen3.6-plus-free","name":"Qwen3.6 Plus Free","description":"Legacy model retained for compatibility with older integrations","family":"qwen-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"ling-3.0-tiny-free":{"id":"ling-3.0-tiny-free","name":"Ling-3.0-tiny Free","description":"Compact MoE model for responsive agents, instruction following, and multi-turn conversations","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0}}}},"moonshotai-cn":{"id":"moonshotai-cn","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.cn/v1","name":"Moonshot AI (China)","doc":"https://platform.moonshot.cn/docs/api/chat","models":{"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}}}},"stepfun-step-plan":{"id":"stepfun-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/step_plan/v1","name":"StepFun Step Plan (China)","doc":"https://platform.stepfun.com/docs/zh/step-plan/integrations/reasoning-api","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-router-v1":{"id":"step-router-v1","name":"Step Router v1","description":"StepFun routing model that dispatches requests to the appropriate Step model.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":256000}}}},"nearai":{"id":"nearai","env":["NEARAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://cloud-api.near.ai/v1","name":"NEAR AI Cloud","doc":"https://docs.near.ai/","models":{"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.4,"output":4.4}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen 3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.17,"output":1.1,"cache_read":0.056}},"Qwen/Qwen3-Embedding-0.6B":{"id":"Qwen/Qwen3-Embedding-0.6B","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3-Reranker-0.6B":{"id":"Qwen/Qwen3-Reranker-0.6B","name":"Qwen3 Reranker 0.6B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen3-VL 30B-A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.15,"output":0.55}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.01,"output":0.01}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"black-forest-labs/FLUX.2-klein-4B":{"id":"black-forest-labs/FLUX.2-klein-4B","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":1,"output":1}}}},"openrouter":{"id":"openrouter","env":["OPENROUTER_API_KEY"],"npm":"@openrouter/ai-sdk-provider","api":"https://openrouter.ai/api/v1","name":"OpenRouter","doc":"https://openrouter.ai/models","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.475,"output":4.425,"cache_read":0.295,"cache_write":1.84375}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125,"tiers":[{"input":1.17,"output":5.85,"cache_read":0.234,"cache_write":1.4625,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":1.1}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375,"tiers":[{"input":0.325,"output":1.625,"cache_read":0.065,"cache_write":0.40625,"tier":{"type":"context","size":32000}},{"input":0.52,"output":2.6,"cache_read":0.104,"cache_write":0.65,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.12,"output":0.24}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625,"tiers":[{"input":1.3,"output":3.9,"cache_write":1.625,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.3,"output":3.9,"cache_write":1.625}}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375,"tiers":[{"input":0.375,"output":2.25,"cache_write":0.46875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.375,"output":2.25,"cache_write":0.46875}}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56,"tiers":[{"input":0.325,"output":1.95,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.325,"output":1.95}}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"tiers":[{"input":0.78,"output":2.34,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34}}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.12,"output":0.8,"cache_read":0.07}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.07,"output":0.28}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.0875,"output":0.35,"cache_read":0.0175}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.55,"output":3.5,"cache_read":0.225}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2,"cache_read":0.03}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"cache_write":0.125,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"cache_write":0.25,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"tiers":[{"input":1.56,"output":7.8,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975,"tiers":[{"input":1.56,"output":7.8,"cache_read":0.312,"cache_write":1.95,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.52}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325,"tiers":[{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975}}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen3.8-27b:free":{"id":"qwen/qwen3.8-27b:free","name":"Qwen3.8 27B (free)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.04815,"output":0.19305}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375,"tiers":[{"input":0.75,"output":3,"cache_write":0.9375,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":3,"cache_write":0.9375}}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.12,"output":0.5}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375,"tiers":[{"input":1.58,"output":9.48,"cache_write":1.975,"tier":{"type":"context","size":128000}}]}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2}}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B ","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"Aion-3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"Aion-3.0-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":943718},"cost":{"input":0.04,"output":0.2,"cache_read":0.01}},"~deepseek/deepseek-pro-latest":{"id":"~deepseek/deepseek-pro-latest","name":"DeepSeek Pro Latest","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.5544,"output":1.6632,"cache_read":0.01848}},"~deepseek/deepseek-flash-latest":{"id":"~deepseek/deepseek-flash-latest","name":"DeepSeek Flash Latest","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.12,"output":0.48,"cache_read":0.0036}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots3-Note Preview (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"LongCat 2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048756,"output":262144},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"Ternary Bonsai 2 27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.5}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.12,"cache_read":0.03}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Laguna XS 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Laguna S 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.09,"output":0.18,"cache_read":0.009}},"kwaipilot/kat-coder-pro-v2":{"id":"kwaipilot/kat-coder-pro-v2","name":"KAT-Coder-Pro V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":144000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"KAT-Coder-Pro V2.5","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11-30","release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-03-31","release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-01-31","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.6-pro-ultraspeed":{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","name":"MiMo-V2.6-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.255,"output":1.02}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.27,"output":1.08,"cache_read":0.027}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.4,"output":2.2}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-03-31","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000192,"output":900172},"cost":{"input":0.2,"output":1.1}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"Nemotron 3.5 Lightning (free)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.07,"output":0.2,"cache_read":0.04}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.08,"output":0.45}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"Nemotron 3.5 Content Safety (free)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":182520},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.09,"output":0.3,"cache_read":0.05}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.45,"cache_read":0.04}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemma-4-31b-it:free":{"id":"google/gemma-4-31b-it:free","name":"Gemma 4 31B (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemma-4-26b-a4b-it:free":{"id":"google/gemma-4-26b-a4b-it:free","name":"Gemma 4 26B A4B (free)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex-N2.5-Mini (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex-N2.5-Pro (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Inkling Small (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling:free":{"id":"thinkingmachines/inkling:free","name":"Inkling (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":471859},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":3686},"cost":{"input":0.08,"output":0.11}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3,"tiers":[{"input":0.1,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3,"tiers":[{"input":1,"output":6,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4,"tiers":[{"input":0.2,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Fugu Ultra v2","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-08-28","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.7,"output":8.5,"cache_read":0.17}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943718},"cost":{"input":0.04,"output":0.32,"cache_read":0.016}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.05544,"output":0.11088,"cache_read":0.011088}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.32,"output":0.89}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.91263,"output":1.82526,"cache_read":0.076053}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":147456},"cost":{"input":0.25,"output":1}},"~openai/gpt-terra-latest":{"id":"~openai/gpt-terra-latest","name":"GPT Terra Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"~openai/gpt-sol-latest":{"id":"~openai/gpt-sol-latest","name":"GPT Sol Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"~openai/gpt-luna-latest":{"id":"~openai/gpt-luna-latest","name":"GPT Luna Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"~openai/gpt-astra-latest":{"id":"~openai/gpt-astra-latest","name":"GPT Astra Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.021,"output":0.063,"cache_read":0.0042}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"Ling 3.0 Flash Fin (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"Ling 3.0 Flash Sante (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl:free":{"id":"inclusionai/ling-3.0-flash-vl:free","name":"Ling 3.0 Flash VL (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"mancer/weaver":{"id":"mancer/weaver","name":"Weaver (alpha)","description":"General-purpose chat model for instruction following, writing, and analysis","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"openrouter/free":{"id":"openrouter/free","name":"Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":8000},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":200000}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"openrouter/fusion":{"id":"openrouter/fusion","name":"Fusion","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-11-08","last_updated":"2023-11-08","modalities":{"input":["text","image","audio","pdf","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.08,"cache_read":0.025}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.2,"output":0.8}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-06-30","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-10-31","release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.03,"output":0.13,"cache_read":0.03}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"GPT-5 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"GLM Flash Latest","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":102400},"cost":{"input":0.075,"output":0.25,"cache_read":0.02}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"GLM Latest","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":131072},"cost":{"input":0.7728,"output":2.4288,"cache_read":0.14352}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.7062,"output":3.21,"cache_read":0.18}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"North Mini Code (free)","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.09,"output":0.36,"cache_read":0.018}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Hy-MT2-30B-A3B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Hy-MT2-7B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Hy-MT2-1.8B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LFM2.5-2.6B (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.6496,"output":2.0416,"cache_read":0.12064}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":102400},"cost":{"input":0.075,"output":0.25,"cache_read":0.02}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5.2:free":{"id":"z-ai/glm-5.2:free","name":"GLM 5.2 (free)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0,"output":0}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.966,"output":3.036,"cache_read":0.1794}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":131072},"cost":{"input":0.91,"output":2.86,"cache_read":0.169}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":117964},"cost":{"input":0.0605,"output":0.4}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Uncensored","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}}}},"cline-pass":{"id":"cline-pass","env":["CLINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cline.bot/api/v1","name":"ClinePass","doc":"https://docs.cline.bot/getting-started/clinepass","models":{"cline-pass/qwen3.7-max":{"id":"cline-pass/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"cline-pass/kimi-k2.6":{"id":"cline-pass/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"cline-pass/glm-5.2":{"id":"cline-pass/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/minimax-m3":{"id":"cline-pass/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"cline-pass/deepseek-v4-flash":{"id":"cline-pass/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/kimi-k2.7-code":{"id":"cline-pass/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"cline-pass/deepseek-v4.1-flash":{"id":"cline-pass/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"cline-pass/kimi-k3":{"id":"cline-pass/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"cline-pass/glm-5.3-flash":{"id":"cline-pass/glm-5.3-flash","name":"cline-pass/glm-5.3-flash","description":"Latest natively multimodal model in the GLM-5 series","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"cline-pass/qwen3.8-max":{"id":"cline-pass/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"cline-pass/qwen3.7-plus":{"id":"cline-pass/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"cline-pass/deepseek-v4-pro":{"id":"cline-pass/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}},"cline-pass/glm-5.3":{"id":"cline-pass/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/mimo-v2.5":{"id":"cline-pass/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/mimo-v2.5-pro":{"id":"cline-pass/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}}}},"iteracompute":{"id":"iteracompute","env":["ITERACOMPUTE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.iteracompute.com/v1","name":"IteraCompute","doc":"https://iteracompute.com/docs.html","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":970000,"output":131072},"cost":{"input":1.95,"output":5.95,"cache_read":0.2}},"ornith-ai/ornith-1.5-35b-a3b":{"id":"ornith-ai/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":3,"cache_read":0.03}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":524288},"cost":{"input":0.29,"output":1.2,"cache_read":0.08}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.1,"output":3.3,"cache_read":0.11}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":970000,"output":393216},"cost":{"input":0.34,"output":1.05,"cache_read":0.035}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":999999},"cost":{"input":3,"output":14.9,"cache_read":0.29}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.49,"cache_read":0.03}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":3.5,"cache_read":0.26}}}},"model-oracle-ai":{"id":"model-oracle-ai","env":["MODEL_ORACLE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.modeloracle.com/api/v1","name":"Model Oracle AI","doc":"https://modeloracle.com/setup/","models":{"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"auto":{"id":"auto","name":"Auto","description":"Model Oracle AI decision engine that selects and routes among configured coding-agent models","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-29","last_updated":"2026-07-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}}}},"ofox":{"id":"ofox","env":["OFOX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ofox.ai/v1","name":"Ofox","doc":"https://ofox.ai/docs","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.8,"output":9,"cache_read":0.2,"cache_write":1}},"qwen/qwen-vl-max":{"id":"qwen/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.23,"output":0.58,"cache_read":0.023}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":2.5,"cache_read":0.06,"cache_write":0.27}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8000},"cost":{"input":0.35,"output":1.38,"cache_read":0.069}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.5,"output":1.71,"cache_read":0.043,"cache_write":0.63}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.022,"output":0.22,"cache_read":0.0043,"cache_write":0.027}},"qwen/qwen-turbo":{"id":"qwen/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.043,"output":0.09,"cache_read":0.0086}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.43,"output":2.57}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.36,"output":1.43,"cache_read":0.072}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.29,"cache_read":0.29}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11,"output":0.39,"cache_read":0.011,"cache_write":0.14}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"bailian/qwen3.7-max":{"id":"bailian/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"bailian/qwen3-coder-plus":{"id":"bailian/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":9,"cache_read":0.2,"cache_write":1}},"bailian/qwen-vl-max":{"id":"bailian/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.58,"cache_read":0.046}},"bailian/qwen3-coder-flash":{"id":"bailian/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.06,"cache_write":0.27}},"bailian/qwen-max":{"id":"bailian/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.35,"output":1.38,"cache_read":0.069}},"bailian/qwen3.6-plus":{"id":"bailian/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"bailian/qwen3.5-27b":{"id":"bailian/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"bailian/qwen3.8-27b":{"id":"bailian/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.05,"cache_write":0.5625}},"bailian/qwen3.5-35b-a3b":{"id":"bailian/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"bailian/qwen-flash":{"id":"bailian/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.22,"cache_read":0.0043,"cache_write":0.027}},"bailian/qwen-turbo":{"id":"bailian/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.05,"output":0.09,"cache_read":0.0086}},"bailian/qwen3.5-flash":{"id":"bailian/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"bailian/qwen3-coder-next":{"id":"bailian/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"bailian/qwen3.5-397b-a17b":{"id":"bailian/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"bailian/qwen3.6-27b":{"id":"bailian/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.6,"output":3.6}},"bailian/qwen3.8-max-0902":{"id":"bailian/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"bailian/qwen3-max":{"id":"bailian/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.36,"output":1.43,"cache_read":0.072}},"bailian/qwen-plus":{"id":"bailian/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"bailian/qwen3.5-122b-a10b":{"id":"bailian/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.29,"cache_read":0.29}},"bailian/qwen3.6-flash":{"id":"bailian/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"bailian/qwen3.8-flash":{"id":"bailian/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"bailian/qwen3.6-max-preview":{"id":"bailian/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"bailian/qwen3.8-max":{"id":"bailian/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"bailian/qwen3.7-plus":{"id":"bailian/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"bailian/qwen3.5-plus":{"id":"bailian/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"volcengine/doubao-seed-2.1-turbo":{"id":"volcengine/doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3536,"output":1.7696,"cache_read":0.068,"cache_write":0.0019}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.06,"output":0.56,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-2.1-pro":{"id":"volcengine/doubao-seed-2.1-pro","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.7072,"output":3.536,"cache_read":0.1416,"cache_write":0.002}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"volcengine/doubao-seed-1-8":{"id":"volcengine/doubao-seed-1-8","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"volcengine/doubao-seed-evolving":{"id":"volcengine/doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.884,"output":4.42,"cache_read":0.177,"cache_write":0.0025}},"volcengine/doubao-seed-character":{"id":"volcengine/doubao-seed-character","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.177,"output":0.884,"cache_read":0.024,"cache_write":0.0025}},"volcengine/doubao-seed-1-6-vision":{"id":"volcengine/doubao-seed-1-6-vision","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.12,"output":1.15,"cache_read":0.023}},"volcengine/doubao-seed-1-6-flash":{"id":"volcengine/doubao-seed-1-6-flash","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.03,"output":0.22,"cache_read":0.0043}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.13,"output":0.76,"cache_read":0.03,"cache_write":0.0024}},"volcengine/doubao-seed-1-6":{"id":"volcengine/doubao-seed-1-6","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax-M2.1 Lightning","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/m2-her":{"id":"minimax/m2-her","name":"MiniMax-M2 Her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax-M2.5 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":1,"input_audio":0.3}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083,"input_audio":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":0.75}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":1,"input_audio":1}},"deepseek/deepseek-v4-flash-0423":{"id":"deepseek/deepseek-v4-flash-0423","name":"DeepSeek V4 Flash 0423","description":"Initial DeepSeek V4 Flash snapshot for economical reasoning, coding, and million-token agent workloads","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.19,"output":0.51,"cache_read":0.028}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"beta","cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.924,"output":2.772,"cache_read":0.0308}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.308,"output":0.924,"cache_read":0.0098}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.21,"output":0.84,"cache_read":0.0042}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.29,"output":0.43,"cache_read":0.06}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.15}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":4,"output":12,"cache_read":0.4}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"Grok 4.1 Fast","description":"xAI's fast agentic tool-calling model with a 2M context window; non-reasoning variant for low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.04,"output":0.32,"cache_read":0.008}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.6,"cache_read":0.024}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.32,"output":1.28,"cache_read":0.08}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":12,"cache_read":0.2}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":8,"cache_read":1}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.12,"output":0.48,"cache_read":0.06}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.6,"output":6.4,"cache_read":0.4}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.16,"output":1,"cache_read":0.016}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.6,"output":3.6,"cache_read":0.06}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.6,"cache_read":0.024}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":24,"output":144}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":4,"output":24,"cache_read":0.4}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.072,"output":0.4,"cache_read":0.01}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}}}},"arcee":{"id":"arcee","env":["ARCEE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.arcee.ai/api/v1","name":"Arcee","doc":"https://docs.arcee.ai","models":{"trinity-large-thinking":{"id":"trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"status":"beta","cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":3,"output":15,"cache_read":0.3}}}},"kuae-cloud-coding-plan":{"id":"kuae-cloud-coding-plan","env":["KUAE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-plan-endpoint.kuaecloud.net/v1","name":"KUAE Cloud Coding Plan","doc":"https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/","models":{"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"ebcloud":{"id":"ebcloud","env":["EBCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://maas-api.ebcloud.com/v1","name":"EBCloud","doc":"https://docs.ebtech.com/ai/model-api.html","models":{"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.143,"output":0.2857}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.8571,"output":3.4286}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9286,"output":3.8571}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4286,"output":0.8571}}}},"agnes":{"id":"agnes","env":["AGNES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apihub.agnes-ai.com/v1","name":"Agnes AI","doc":"https://agnes-ai.com/doc","models":{"agnes-2.5-pro-alpha":{"id":"agnes-2.5-pro-alpha","name":"Agnes 2.5 Pro Alpha","description":"Paid reasoning model for advanced coding, scientific reasoning, long-context analysis, agentic workflows, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.45,"output":0.9,"cache_read":0.0038}},"agnes-2.5-flash":{"id":"agnes-2.5-flash","name":"Agnes 2.5 Flash","description":"Upgraded model with improved coding, agent workflows, tool calling, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07","last_updated":"2026-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}},"agnes-2.0-flash":{"id":"agnes-2.0-flash","name":"Agnes 2.0 Flash","description":"Fast and efficient model for agent workflows, tool calling, coding, and image understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-25","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}}}},"amd":{"id":"amd","env":["AMD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://developer.amd.com.cn/radeon/api/v1","name":"AMD","doc":"https://developer.amd.com.cn/radeon/tokenfactory","models":{"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"Qwen3.8-Flash-Next":{"id":"Qwen3.8-Flash-Next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"DeepSeek-V4-Flash-Vision-Exp":{"id":"DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"MiniCPM5-2B":{"id":"MiniCPM5-2B","name":"MiniCPM5-2B","description":"Dense 2B-class open-source model for on-device and resource-constrained use, with native long-context support, tool calling, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-09-06","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.124,"output":0.7425,"cache_read":0.124}},"DeepSeek-V4.1-Flash":{"id":"DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}}}},"xiaomi-token-plan-sgp":{"id":"xiaomi-token-plan-sgp","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-sgp.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Singapore)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}}}},"neon":{"id":"neon","env":["NEON_AI_GATEWAY_BASE_URL","NEON_AI_GATEWAY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"${NEON_AI_GATEWAY_BASE_URL}/v1","name":"Neon","doc":"https://neon.com/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen35-122b-a10b":{"id":"qwen35-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":25000},"cost":{"input":0.22,"output":2.2}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":10000},"cost":{"input":0.15,"output":1.2}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"meta-llama-3-3-70b-instruct":{"id":"meta-llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.07,"output":0.3}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5-2":{"id":"gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.5,"output":1.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.3}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"meta-llama-3-1-8b-instruct":{"id":"meta-llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Meta's compact open-weight Llama 3.1 model for fast, low-cost text generation","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.45}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5-1":{"id":"gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5-5-pro":{"id":"gpt-5-5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":524288},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2,"output":6,"cache_read":0.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.15,"output":0.6}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemma-3-12b":{"id":"gemma-3-12b","name":"Gemma 3 12B","description":"Google's open-weight Gemma 3 vision-language model for text and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.5}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gpt-5-4":{"id":"gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}}}},"qihang-ai":{"id":"qihang-ai","env":["QIHANG_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qhaigc.net/v1","name":"QiHang","doc":"https://www.qhaigc.net/docs","models":{"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.14,"output":1.14}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.57,"output":3.43}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.43,"output":2.14}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.14,"output":0.71}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.07,"output":0.43,"tiers":[{"input":0.07,"output":0.43,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.07,"output":0.43}}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.04,"output":0.29}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.71,"tiers":[{"input":0.09,"output":0.71,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.09,"output":0.71}}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":0.71,"output":3.57}}}},"scnet-token-plan":{"id":"scnet-token-plan","env":["SCNET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scnet.cn/api/llm/v1","name":"SCNet Token Plan","doc":"https://www.scnet.cn/ac/openapi/doc/2.0/moduleapi/plans/token-plan.html","models":{"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Flash-0731":{"id":"DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Flash":{"id":"Qwen3.8-Flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3":{"id":"GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5":{"id":"GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4.1-Flash":{"id":"DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro-0813":{"id":"DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3-Flash":{"id":"GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"inference":{"id":"inference","env":["INFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.net/v1","name":"Inference","doc":"https://inference.net/models","models":{"qwen/qwen-2.5-7b-vision-instruct":{"id":"qwen/qwen-2.5-7b-vision-instruct","name":"Qwen 2.5 7B Vision Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.2,"output":0.2}},"qwen/qwen3-embedding-4b":{"id":"qwen/qwen3-embedding-4b","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"google/gemma-3":{"id":"google/gemma-3","name":"Google Gemma 3","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.15,"output":0.3}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.025,"output":0.025}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.02,"output":0.02}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.01,"output":0.01}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.055,"output":0.055}},"osmosis/osmosis-structure-0.6b":{"id":"osmosis/osmosis-structure-0.6b","name":"Osmosis Structure 0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"osmosis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":2048},"cost":{"input":0.1,"output":0.5}},"mistral/mistral-nemo-12b-instruct":{"id":"mistral/mistral-nemo-12b-instruct","name":"Mistral Nemo 12B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.038,"output":0.1}}}},"openai":{"id":"openai","env":["OPENAI_API_KEY"],"npm":"@ai-sdk/openai","name":"OpenAI","doc":"https://platform.openai.com/docs/models","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-4o-2024-05-13":{"id":"gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":5,"output":15}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"chatgpt-image-latest":{"id":"chatgpt-image-latest","name":"chatgpt-image-latest","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"gpt-4o-2024-08-06":{"id":"gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":100000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"o1-pro":{"id":"o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":150,"output":600}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2022-12","release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-image-1":{"id":"gpt-image-1","name":"gpt-image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"status":"deprecated"},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-image-1-mini":{"id":"gpt-image-1-mini","name":"gpt-image-1-mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-image-2":{"id":"gpt-image-2","name":"gpt-image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"status":"deprecated","cost":{"input":0.5,"output":1.5,"cache_read":0}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":30,"output":60}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"gpt-realtime-2.1":{"id":"gpt-realtime-2.1","name":"GPT-Realtime-2.1","description":"Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4,"input_audio":32,"output_audio":64}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}}}},"aiand":{"id":"aiand","env":["AIAND_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aiand.com/v1","name":"ai&","doc":"https://docs.aiand.com/","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3,"cache_read":0.2}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2,"cache_read":0.2}},"motif-technologies/motif-3":{"id":"motif-technologies/motif-3","name":"Motif 3","description":"Motif 3 is a large-scale, decoder-only Mixture-of-Experts (MoE) language model with 314 billion total parameters and 13.2 billion parameters activated per token.","family":"motif","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2,"cache_read":0.2}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.25,"cache_read":0.08}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1,"output":2.5,"cache_read":0.25}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4,"cache_read":0.3}},"zai-org/glm-5.3":{"id":"zai-org/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4,"cache_read":0.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.08}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5,"cache_read":0.2}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":12.5,"cache_read":0.5}}}},"siliconflow":{"id":"siliconflow","env":["SILICONFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.com/v1","name":"SiliconFlow","doc":"https://cloud.siliconflow.com/models","models":{"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.28,"cache_read":0.028}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.014}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"deepseek-ai/DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"deepseek-ai/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.41}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42,"cache_read":0.135}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.50162,"output":3.135,"cache_read":0.135}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.4}},"google/gemma-4-12B-it":{"id":"google/gemma-4-12B-it","name":"Gemma 4 12B IT","description":"Compact Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":131000},"cost":{"input":1.19,"output":3.74,"cache_read":0.6,"cache_write":0}},"zai-org/GLM-5V-Turbo":{"id":"zai-org/GLM-5V-Turbo","name":"zai-org/GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.302,"output":4.092,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":0.95,"output":2.55,"cache_read":0.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"nex-agi/Nex-N2-Pro":{"id":"nex-agi/Nex-N2-Pro","name":"Nex-N2-Pro","description":"Open agentic MoE model (397B total, 17B active) for coding, tool use, and research workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.08}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":131000},"cost":{"input":2,"output":6,"cache_read":0.25}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.39,"output":2.34}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.24,"output":1.8}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":3.2}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":1.6}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"meituan-longcat/LongCat-2.0":{"id":"meituan-longcat/LongCat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1049000,"output":131072},"cost":{"input":0.75,"output":2.95,"cache_read":0.015}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMaxAI/MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":197000,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"openai/gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.04,"output":0.18}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"openai/gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.05,"output":0.45}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.85916,"output":3.8,"cache_read":0.17993}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.77,"output":3.4,"cache_read":0.14}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262000},"cost":{"input":2.7,"output":13.5,"cache_read":0.27}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":262144},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}}}},"stepfun-ai-step-plan":{"id":"stepfun-ai-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/step_plan/v1","name":"StepFun Step Plan (Global)","doc":"https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}}}},"hetzner":{"id":"hetzner","env":["HETZNER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.hetzner.com/api/v1","name":"Hetzner","doc":"https://experiments.hetzner.com/docs/inference","models":{"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}}}},"snowflake-cortex":{"id":"snowflake-cortex","env":["SNOWFLAKE_ACCOUNT","SNOWFLAKE_CORTEX_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1","name":"Snowflake Cortex","doc":"https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"openai-gpt-5.1":{"id":"openai-gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"mistral-large2":{"id":"mistral-large2","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"openai-gpt-5":{"id":"openai-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta"},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta"},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192},"status":"beta"},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"snowflake-llama3.3-70b":{"id":"snowflake-llama3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096}}}},"meganova":{"id":"meganova","env":["MEGANOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.meganova.ai/v1","name":"Meganova","doc":"https://docs.meganova.ai","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":0.88}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.4}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.26,"output":0.38}},"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.02,"output":0.04}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.2,"output":0.8}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.8,"output":2.56}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.45,"output":1.9}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.5-Plus":{"id":"Qwen/Qwen3.5-Plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.6}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.28,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.3}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.6}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.8}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo V2 Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3}}}},"melious":{"id":"melious","env":["MELIOUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.melious.ai/v1","name":"Melious","doc":"https://melious.ai/docs/reference/models","models":{"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.1592,"output":3.4776,"cache_read":0.11592}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.11592,"output":0.2898,"cache_read":0.023184}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.81144,"output":4.0572,"cache_read":0.266616}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1592,"output":4.6368,"cache_read":0.2898}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.81144,"output":3.4776,"cache_read":0.220248}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.23184,"output":1.1592,"cache_read":0.011592}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3.1878,"output":15.939,"cache_read":0.788256}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":32768},"cost":{"input":0.69552,"output":2.78208,"cache_read":0.185472}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":64000},"cost":{"input":0.34776,"output":0.5796,"cache_read":0.092736}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11592,"output":0.46368,"cache_read":0.023184}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":1.10124,"output":3.36168,"cache_read":0.266616}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.5796,"output":2.95596,"cache_read":0.139104}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":1.50696,"output":4.6368,"cache_read":0.370944}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.85472,"output":3.70944,"cache_read":0.46368}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1592,"output":3.4776,"cache_read":0.23184}}}},"moonshotai":{"id":"moonshotai","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.ai/v1","name":"Moonshot AI","doc":"https://platform.moonshot.ai/docs/api/chat","models":{"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"volcengine-coding-plan":{"id":"volcengine-coding-plan","env":["ARK_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/coding/v3","name":"Volcengine Ark Coding Plan","doc":"https://www.volcengine.com/docs/82379/1928261","models":{"doubao-seed-2.1-turbo":{"id":"doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"302ai":{"id":"302ai","env":["302AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.302.ai/v1","name":"302.AI","doc":"https://doc.302.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"claude-sonnet-4-6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.33,"output":0.33}},"glm-4.7":{"id":"glm-4.7","name":"glm-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"gemini-3.5-flash-thinking":{"id":"gemini-3.5-flash-thinking","name":"gemini-3.5-flash-thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4}},"claude-sonnet-4-6-thinking":{"id":"claude-sonnet-4-6-thinking","name":"claude-sonnet-4-6-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-10-26","last_updated":"2025-10-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.33,"output":1.32}},"glm-4.6":{"id":"glm-4.6","name":"glm-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"gemini-2.5-flash-image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.145,"output":0.43}},"deepseek-v3.2-thinking":{"id":"deepseek-v3.2-thinking","name":"DeepSeek-V3.2-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.29,"output":0.43}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.06,"output":0.46}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50}},"gpt-5.6-luna-pro":{"id":"gpt-5.6-luna-pro","name":"gpt-5.6-luna-pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.575,"output":2.3}},"claude-sonnet-4-5-20250929-thinking":{"id":"claude-sonnet-4-5-20250929-thinking","name":"claude-sonnet-4-5-20250929-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"qwen3.7-max-2026-06-08":{"id":"qwen3.7-max-2026-06-08","name":"qwen3.7-max-2026-06-08","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"qwen3-max-2025-09-23":{"id":"qwen3-max-2025-09-23","name":"qwen3-max-2025-09-23","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":258048,"output":65536},"cost":{"input":0.86,"output":3.43}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"gemini-2.0-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-11","release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":8192},"cost":{"input":0.075,"output":0.3}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":0,"tiers":[{"input":5,"output":22.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"gpt-5.6-sol-pro":{"id":"gpt-5.6-sol-pro","name":"gpt-5.6-sol-pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"gemini-2.5-flash-preview-09-2025","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"claude-opus-4-7-thinking":{"id":"claude-opus-4-7-thinking","name":"claude-opus-4-7-thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"gpt-5.1-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"mistral-large-2512":{"id":"mistral-large-2512","name":"mistral-large-2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":1.1,"output":3.3}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gpt-4o":{"id":"gpt-4o","name":"gpt-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"claude-opus-4-7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.29,"output":0.43}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.283,"output":1.705}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.286,"output":1.142}},"kimi-k2-0905-preview":{"id":"kimi-k2-0905-preview","name":"kimi-k2-0905-preview","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.632,"output":2.53}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"doubao-seed-1-8-251215":{"id":"doubao-seed-1-8-251215","name":"doubao-seed-1-8-251215","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":224000,"output":64000},"cost":{"input":0.114,"output":0.286}},"grok-4.1":{"id":"grok-4.1","name":"grok-4.1","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2,"output":10}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"gpt-5.4-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"gpt-5.6-terra-pro":{"id":"gpt-5.6-terra-pro","name":"gpt-5.6-terra-pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"gemini-3-pro-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":64000},"cost":{"input":2,"output":120}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax-M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.132,"output":1.254}},"gemini-2.5-flash-nothink":{"id":"gemini-2.5-flash-nothink","name":"gemini-2.5-flash-nothink","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-24","last_updated":"2025-06-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.29,"output":0.86}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.188,"output":1.133}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":1.08}},"gpt-5-thinking":{"id":"gpt-5-thinking","name":"gpt-5-thinking","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.18,"output":0.564}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"gpt-5.4-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"claude-haiku-4-5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"glm-5":{"id":"glm-5","name":"glm-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.6}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.16,"output":6.36}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.3}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.29,"output":2.86}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.285,"output":1.15}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"qwen3-235b-a22b-instruct-2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":0.29,"output":1.143}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.72,"output":2.88}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"glm-5-turbo":{"id":"glm-5-turbo","name":"glm-5-turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"qwen3-coder-480b-a35b-instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.86,"output":3.43}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":10}},"claude-opus-5-thinking":{"id":"claude-opus-5-thinking","name":"claude-opus-5-thinking","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"gemini-3.1-flash-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"doubao-seed-1-6-vision-250815","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.114,"output":1.143}},"deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"doubao-seed-1-6-thinking-250715":{"id":"doubao-seed-1-6-thinking-250715","name":"doubao-seed-1-6-thinking-250715","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16000},"cost":{"input":0.121,"output":1.21}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"grok-4.20-beta-0309-reasoning":{"id":"grok-4.20-beta-0309-reasoning","name":"grok-4.20-beta-0309-reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"gpt-5.2-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"claude-opus-4-1-20250805-thinking":{"id":"claude-opus-4-1-20250805-thinking","name":"claude-opus-4-1-20250805-thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-27","last_updated":"2025-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.12,"output":0.69}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}}}},"cohere":{"id":"cohere","env":["COHERE_API_KEY"],"npm":"@ai-sdk/cohere","name":"Cohere","doc":"https://docs.cohere.com/docs/models","models":{"command-r7b-arabic-02-2025":{"id":"command-r7b-arabic-02-2025","name":"Command R7B Arabic","description":"Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"command-a-plus-05-2026":{"id":"command-a-plus-05-2026","name":"Command A Plus","description":"Cohere's stronger command model for multilingual agents and enterprise workflows","family":"command-a","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04-01","release_date":"2026-05-20","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":2.5,"output":10}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Command A Reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":2.5,"output":10}},"command-a-vision-07-2025":{"id":"command-a-vision-07-2025","name":"Command A Vision","description":"Cohere vision model for multilingual document analysis, OCR, and image understanding","family":"command-a","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":2.5,"output":10}},"north-mini-code-1-0":{"id":"north-mini-code-1-0","name":"North Mini Code","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.cohere.ai/compatibility/v1"},"cost":{"input":0,"output":0}},"command-r-plus-08-2024":{"id":"command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"command-a-translate-08-2025":{"id":"command-a-translate-08-2025","name":"Command A Translate","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":2.5,"output":10}},"command-a-03-2025":{"id":"command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"c4ai-aya-expanse-32b":{"id":"c4ai-aya-expanse-32b","name":"Aya Expanse 32B","description":"Open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}},"c4ai-aya-expanse-8b":{"id":"c4ai-aya-expanse-8b","name":"Aya Expanse 8B","description":"Compact open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":4000}},"c4ai-aya-vision-8b":{"id":"c4ai-aya-vision-8b","name":"Aya Vision 8B","description":"Compact open multilingual vision model for OCR and visual question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"command-r7b-12-2024":{"id":"command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"c4ai-aya-vision-32b":{"id":"c4ai-aya-vision-32b","name":"Aya Vision 32B","description":"Open multilingual vision model for OCR, visual reasoning, and image question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"command-r-08-2024":{"id":"command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}}}},"upstage":{"id":"upstage","env":["UPSTAGE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.upstage.ai/v1/solar","name":"Upstage","doc":"https://developers.upstage.ai/docs/apis/chat","models":{"solar-mini":{"id":"solar-mini","name":"solar-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"solar-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-06-12","last_updated":"2025-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.15,"output":0.15}},"solar-pro4":{"id":"solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"solar-pro3":{"id":"solar-pro3","name":"solar-pro3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.25,"output":0.25}},"solar-pro2":{"id":"solar-pro2","name":"solar-pro2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.25,"output":0.25}}}},"inco":{"id":"inco","env":["INCO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inco.ai/v1","name":"Inco","doc":"https://platform.inco.ai/docs","models":{"kimi-k3:fast":{"id":"kimi-k3:fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":6,"output":30}},"deepseek-v4.1-flash:fast":{"id":"deepseek-v4.1-flash:fast","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.6,"output":2.4}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2}},"glm-5.3:fast":{"id":"glm-5.3:fast","name":"GLM-5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.8,"output":8.8}},"minimax-m3:fast":{"id":"minimax-m3:fast","name":"MiniMax M3 Fast","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4}},"glm-5.3-flash:fast":{"id":"glm-5.3-flash:fast","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}}}},"sarvam":{"id":"sarvam","env":["SARVAM_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sarvam.ai/v1","name":"Sarvam AI","doc":"https://docs.sarvam.ai/api-reference-docs/getting-started/models","models":{"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam-105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}},"sarvam-30b":{"id":"sarvam-30b","name":"Sarvam-30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536}}}},"xai":{"id":"xai","env":["XAI_API_KEY"],"npm":"@ai-sdk/xai","name":"xAI","doc":"https://docs.x.ai/docs/models","models":{"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.20-0309-reasoning":{"id":"grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.20-multi-agent-0309":{"id":"grok-4.20-multi-agent-0309","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-imagine-image":{"id":"grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-imagine-video":{"id":"grok-imagine-video","name":"Grok Imagine Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"grok-imagine-video-1.5":{"id":"grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Video model for image-to-video generation, editing, and extension workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text","image","audio","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"grok-imagine-image-quality":{"id":"grok-imagine-image-quality","name":"Grok Imagine Image Quality","description":"Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-4.20-0309-non-reasoning":{"id":"grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}}}},"zenifra":{"id":"zenifra","env":["ZENIFRA_AI_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai.zenifra.com/v1","name":"Zenifra","doc":"https://docs.zenifra.com","models":{"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"provider":{"shape":"completions"},"cost":{"input":0.19,"output":0.48}}}},"zai":{"id":"zai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/paas/v4","name":"Z.AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flashx":{"id":"glm-5.3-flashx","name":"GLM-5.3-FlashX","description":"High-speed GLM-5.3-Flash serving option for coding and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-4.6v-flash":{"id":"glm-4.6v-flash","name":"GLM-4.6V-Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"bailing":{"id":"bailing","env":["BAILING_API_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tbox.cn/api/llm/v1/chat/completions","name":"Bailing","doc":"https://alipaytbox.yuque.com/sxs0ba/ling/intro","models":{"Ring-1T":{"id":"Ring-1T","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}},"Ling-1T":{"id":"Ling-1T","name":"Ling-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}}}},"tencent-tokenhub":{"id":"tencent-tokenhub","env":["TENCENT_TOKENHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://tokenhub.tencentmaas.com/v1","name":"Tencent TokenHub","doc":"https://cloud.tencent.com/document/product/1823/130050","models":{"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"hy3-preview":{"id":"hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"runinfra":{"id":"runinfra","env":["RUNINFRA_GATEWAY_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.runinfra.ai/v1","name":"RunInfra","doc":"https://runinfra.ai/docs","models":{"ornith-ai/Ornith-1.5-35B-A3B":{"id":"ornith-ai/Ornith-1.5-35B-A3B","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Inferact/Qwen3.8-2.4T-A95B-NVFP4":{"id":"Inferact/Qwen3.8-2.4T-A95B-NVFP4","name":"Qwen3.8 2.4T A95B (NVFP4)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.2}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.13,"output":0.27,"cache_read":0.01}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.6,"output":1.9,"cache_read":0.03}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.01}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}}}},"ai-router":{"id":"ai-router","env":["AI_ROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ai-router.dev/v1","name":"AI-ROUTER","doc":"https://ai-router.dev/openai-compatible-api-gateway/","models":{"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"berget":{"id":"berget","env":["BERGET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.berget.ai/v1","name":"Berget.AI","doc":"https://api.berget.ai","models":{"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct 2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.33,"output":0.33}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["audio","image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.275,"output":0.55}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":32768},"cost":{"input":1.54,"output":4.84}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":16384},"cost":{"input":0.29,"output":0.58}},"Qwen/Qwen3.8-27B-FP8":{"id":"Qwen/Qwen3.8-27B-FP8","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.46,"output":3.48}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":32768},"cost":{"input":3,"output":15}}}},"mistral":{"id":"mistral","env":["MISTRAL_API_KEY"],"npm":"@ai-sdk/mistral","name":"Mistral","doc":"https://docs.mistral.ai/getting-started/models/","models":{"pixtral-12b":{"id":"pixtral-12b","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"devstral-small-2507":{"id":"devstral-small-2507","name":"Devstral Small","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"magistral-small":{"id":"magistral-small","name":"Magistral Small","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.5,"output":1.5}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral-embed":{"id":"mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":3072},"cost":{"input":0.1,"output":0}},"devstral-small-2505":{"id":"devstral-small-2505","name":"Devstral Small 2505","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"labs-devstral-small-2512":{"id":"labs-devstral-small-2512","name":"Devstral Small 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0}},"magistral-medium-latest":{"id":"magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"zai-glm-5-3":{"id":"zai-glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"open-mixtral-8x22b":{"id":"open-mixtral-8x22b","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":2,"output":6}},"open-mixtral-8x7b":{"id":"open-mixtral-8x7b","name":"Mixtral 8x7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-01","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.7,"output":0.7}},"open-mistral-7b":{"id":"open-mistral-7b","name":"Mistral 7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":0.25,"output":0.25}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"devstral-medium-2507":{"id":"devstral-medium-2507","name":"Devstral Medium","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral-medium-2604":{"id":"mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"devstral-medium-latest":{"id":"devstral-medium-latest","name":"Devstral 2 (latest)","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"voxtral-small-latest":{"id":"voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"ministral-8b-latest":{"id":"ministral-8b-latest","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"voxtral-mini-tts-latest":{"id":"voxtral-mini-tts-latest","name":"Voxtral Mini TTS (latest)","description":"Multilingual text-to-speech model with zero-shot voice cloning","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"mistral-small-latest":{"id":"mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"open-mistral-nemo":{"id":"open-mistral-nemo","name":"Open Mistral Nemo","description":"Legacy model retained for compatibility with older integrations","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"voxtral-mini-latest":{"id":"voxtral-mini-latest","name":"Voxtral Mini (latest)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"devstral-latest":{"id":"devstral-latest","name":"Devstral 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"zai-glm-5-2":{"id":"zai-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"codestral-latest":{"id":"codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"ministral-3b-latest":{"id":"ministral-3b-latest","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"mistral-medium-2508":{"id":"mistral-medium-2508","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"pixtral-large-latest":{"id":"pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}}}},"synthetic":{"id":"synthetic","env":["SYNTHETIC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.synthetic.new/openai/v1","name":"Synthetic","doc":"https://synthetic.new/pricing","models":{"hf:deepseek-ai/DeepSeek-V4.1-Flash":{"id":"hf:deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.6,"output":1.2,"cache_read":0.03}},"hf:openai/gpt-oss-120b":{"id":"hf:openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1}},"hf:MiniMaxAI/MiniMax-M3":{"id":"hf:MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.6,"output":1.2,"cache_read":0.6}},"hf:moonshotai/Kimi-K2.7-Code":{"id":"hf:moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"hf:moonshotai/Kimi-K3":{"id":"hf:moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.45}},"hf:Qwen/Qwen3.6-27B":{"id":"hf:Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.6,"cache_read":0.45}},"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4":{"id":"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.3}},"hf:zai-org/GLM-5.2":{"id":"hf:zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"hf:zai-org/GLM-4.7-Flash":{"id":"hf:zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.1,"output":0.5,"cache_read":0.1}},"hf:zai-org/GLM-5.3-Flash":{"id":"hf:zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}}}},"mixlayer":{"id":"mixlayer","env":["MIXLAYER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.mixlayer.ai/v1","name":"Mixlayer","doc":"https://docs.mixlayer.com","models":{"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.3}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3.2}}}},"longcat":{"id":"longcat","env":["LONGCAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.longcat.chat/openai","name":"LongCat","doc":"https://longcat.chat/platform/docs/","models":{"LongCat-2.0":{"id":"LongCat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.75,"output":2.95,"cache_read":0.015}}}},"cerebras":{"id":"cerebras","env":["CEREBRAS_API_KEY"],"npm":"@ai-sdk/cerebras","name":"Cerebras","doc":"https://inference-docs.cerebras.ai/models/overview","models":{"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.35,"output":0.75}},"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.99,"output":1.49}}}},"togetherai":{"id":"togetherai","env":["TOGETHER_API_KEY"],"npm":"@ai-sdk/togetherai","name":"Together AI","doc":"https://docs.together.ai/docs/serverless-models","models":{"essentialai/Rnj-1-Instruct":{"id":"essentialai/Rnj-1-Instruct","name":"Rnj-1 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"rnj","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"deepseek-ai/DeepSeek-V3-1":{"id":"deepseek-ai/DeepSeek-V3-1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":1.7}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":1.25,"output":1.25}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163839,"output":163839},"status":"deprecated","cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"pearl-ai/gemma-4-31b-it":{"id":"pearl-ai/gemma-4-31b-it","name":"Pearl AI Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.28,"output":0.86}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512300,"output":512300},"cost":{"input":0.6,"output":3.6,"cache_read":0.2}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.39,"output":0.97}},"google/gemma-3n-E4B-it":{"id":"google/gemma-3n-E4B-it","name":"Gemma 3N E4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.06,"output":0.12}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-07","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":164000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1,"output":3.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":400000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2-24B-A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max","xhigh","high","medium","low","none"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":500000},"cost":{"input":1.25,"output":3.75,"cache_read":0.125}},"Qwen/Qwen3-235B-A22B-Instruct-2507-tput":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507-tput","name":"Qwen3 235B A22B Instruct 2507 FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.6-Plus":{"id":"Qwen/Qwen3.6-Plus","name":"Qwen3.6 Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":500000},"cost":{"input":0.5,"output":3}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":130000},"status":"deprecated","cost":{"input":0.6,"output":3.6,"cache_read":0.35}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":2,"output":2}},"Qwen/Qwen2.5-7B-Instruct-Turbo":{"id":"Qwen/Qwen2.5-7B-Instruct-Turbo","name":"Qwen 2.5 7B Instruct Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3}},"Qwen/Qwen3-Coder-Next-FP8":{"id":"Qwen/Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-02-03","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":1.2}},"deepcogito/cogito-v2-1-671b":{"id":"deepcogito/cogito-v2-1-671b","name":"Cogito v2.1 671B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"cogito","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":1.25,"output":1.25}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":250000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":1.04,"output":1.04}},"meta-llama/Meta-Llama-3-8B-Instruct-Lite":{"id":"meta-llama/Meta-Llama-3-8B-Instruct-Lite","name":"Meta Llama 3 8B Instruct Lite","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":2.8}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131000},"cost":{"input":1.2,"output":4.5,"cache_read":0.2}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"cloudflare-workers-ai":{"id":"cloudflare-workers-ai","env":["CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1","name":"Cloudflare Workers AI","doc":"https://developers.cloudflare.com/workers-ai/models/","models":{"@cf/qwen/qwen3-30b-a3b-fp8":{"id":"@cf/qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3b fp8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.0509,"output":0.335}},"@cf/qwen/qwen3.8-27b":{"id":"@cf/qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":3.2,"cache_read":0.05}},"@cf/qwen/qwq-32b":{"id":"@cf/qwen/qwq-32b","name":"Qwq 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.66,"output":1}},"@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.66,"output":1}},"@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"id":"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b","name":"Deepseek R1 Distill Qwen 32B","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.497,"output":4.881}},"@cf/mistralai/mistral-small-3.1-24b-instruct":{"id":"@cf/mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/nvidia/nemotron-3-120b-a12b":{"id":"@cf/nvidia/nemotron-3-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5}},"@cf/google/gemma-4-26b-a4b-it":{"id":"@cf/google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1,"output":0.3}},"@cf/zai-org/glm-5.2":{"id":"@cf/zai-org/glm-5.2","name":"Glm 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/zai-org/glm-5.3-flash":{"id":"@cf/zai-org/glm-5.3-flash","name":"Glm 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"@cf/zai-org/glm-5.3":{"id":"@cf/zai-org/glm-5.3","name":"Glm 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/zai-org/glm-4.7-flash":{"id":"@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma Sea Lion V4 27B It","description":"Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/meta/llama-guard-3-8b":{"id":"@cf/meta/llama-guard-3-8b","name":"Llama Guard 3 8B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.484,"output":0.03}},"@cf/meta/llama-3.1-8b-instruct-fp8":{"id":"@cf/meta/llama-3.1-8b-instruct-fp8","name":"Llama 3.1 8B Instruct fp8","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.152,"output":0.287}},"@cf/meta/llama-3.2-3b-instruct":{"id":"@cf/meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.0509,"output":0.335}},"@cf/meta/llama-3.2-1b-instruct":{"id":"@cf/meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":60000},"cost":{"input":0.027,"output":0.201}},"@cf/meta/llama-4-scout-17b-16e-instruct":{"id":"@cf/meta/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":16384},"cost":{"input":0.27,"output":0.85}},"@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"id":"@cf/meta/llama-3.3-70b-instruct-fp8-fast","name":"Llama 3.3 70B Instruct fp8 Fast","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.293,"output":2.253}},"@cf/meta/llama-3.2-11b-vision-instruct":{"id":"@cf/meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.0485,"output":0.676}},"@cf/ibm-granite/granite-4.0-h-micro":{"id":"@cf/ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 H Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.017,"output":0.112}},"@cf/openai/gpt-oss-20b":{"id":"@cf/openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"@cf/openai/gpt-oss-120b":{"id":"@cf/openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.35,"output":0.75}},"@cf/moonshotai/kimi-k2.6":{"id":"@cf/moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"@cf/moonshotai/kimi-k2.7-code":{"id":"@cf/moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"moark":{"id":"moark","env":["MOARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://moark.com/v1","name":"Moark","doc":"https://moark.com/docs/openapi/v1#tag/%E6%96%87%E6%9C%AC%E7%94%9F%E6%88%90","models":{"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":2.1,"output":8.4,"cache_read":2.1,"cache_write":8.4}},"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":3.5,"output":14}}}},"zenmux":{"id":"zenmux","env":["ZENMUX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://zenmux.ai/api/v1","name":"ZenMux","doc":"https://docs.zenmux.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6-Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1020000,"output":1020000},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3-Max-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":1.2,"output":6}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5}}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.8,"output":4.8}},"baidu/ernie-5.0-thinking-preview":{"id":"baidu/ernie-5.0-thinking-preview","name":"ERNIE 5.0","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.84,"output":3.37}},"volcengine/doubao-seed-code":{"id":"volcengine/doubao-seed-code","name":"Doubao-Seed-Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-11","last_updated":"2025-11-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0.17,"output":1.12,"cache_read":0.03}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Doubao-Seed-2.0-mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.03,"output":0.28,"cache_read":0.01,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.9,"output":4.48}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Doubao-Seed-2.0-pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.45,"output":2.24,"cache_read":0.09,"cache_write":0.0024}},"volcengine/doubao-seed-1.8":{"id":"volcengine/doubao-seed-1.8","name":"Doubao-Seed-1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11,"output":0.28,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Doubao-Seed-2.0-lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.09,"output":0.51,"cache_read":0.02,"cache_write":0.0024}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun/step-3":{"id":"stepfun/step-3","name":"Step-3","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":64000},"status":"deprecated","cost":{"input":0.21,"output":0.57}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash-free":{"id":"stepfun/step-3.7-flash-free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"xiaomi/mimo-v2-pro":{"id":"xiaomi/mimo-v2-pro","name":"MiMo V2 Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":256000},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomi/mimo-v2-omni":{"id":"xiaomi/mimo-v2-omni","name":"MiMo V2 Omni","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":265000,"output":265000},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.611,"output":2.4439}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3055,"output":1.2219}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":2.4}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax M2.5 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":4.8,"cache_read":0.06,"cache_write":0.75}},"anthropic/claude-sonnet-5-free":{"id":"anthropic/claude-sonnet-5-free","name":"Claude Sonnet 5 (Free)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3.5-haiku":{"id":"anthropic/claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2024-11-04","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-3.7-sonnet":{"id":"anthropic/claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":4}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-19","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.03,"cache_write":1}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":65530},"cost":{"input":0.25,"output":1.5}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.3,"output":2.5,"cache_read":0.07,"cache_write":1}},"sapiens-ai/agnes-1.5-lite":{"id":"sapiens-ai/agnes-1.5-lite","name":"Agnes 1.5 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.12,"output":0.6}},"sapiens-ai/agnes-1.5-pro":{"id":"sapiens-ai/agnes-1.5-pro","name":"Agnes 1.5 Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-21","last_updated":"2026-03-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.16,"output":0.8}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek-V3.2 (Non-thinking Mode)","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.28,"output":0.42,"cache_read":0.03}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.28,"output":0.43}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163000,"output":64000},"cost":{"input":0.22,"output":0.33}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"kuaishou/kat-coder-pro-v2":{"id":"kuaishou/kat-coder-pro-v2","name":"KAT-Coder-Pro-V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12-31","release_date":"2026-05-07","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ring-1t":{"id":"inclusionai/ring-1t","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-12","last_updated":"2025-10-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"inclusionai/ling-1t":{"id":"inclusionai/ling-1t","name":"Ling-1T","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.2-fast-non-reasoning":{"id":"x-ai/grok-4.2-fast-non-reasoning","name":"Grok 4.2 Fast Non Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":4,"output":12,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"x-ai/grok-imagine-image-2.0":{"id":"x-ai/grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":66000,"output":0}},"x-ai/grok-voice-stt-1.0":{"id":"x-ai/grok-voice-stt-1.0","name":"Grok Voice STT 1.0","description":"Grok Voice STT 1.0 is xAI's speech-to-text model. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio.","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":15000,"output":15000}},"x-ai/grok-4.2-fast":{"id":"x-ai/grok-4.2-fast","name":"Grok 4.2 Fast","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":4,"output":12,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"x-ai/grok-voice-tts-1.0":{"id":"x-ai/grok-voice-tts-1.0","name":"Grok Voice TTS 1.0","description":"Convert text into spoken audio with a single API call. The API supports a rich set of expressive voices, inline speech tags for fine-grained delivery control, and output formats from high-fidelity MP3 to telephony-optimized μ-law.","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":15000,"output":15000}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-15","last_updated":"2026-01-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":21,"output":168}},"openai/gpt-5.1-chat":{"id":"openai/gpt-5.1-chat","name":"GPT-5.1 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":3.75,"output":18.75}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.2,"output":1.25}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.75,"output":4.5}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":45,"output":225}},"openai/gpt-5.5-instant":{"id":"openai/gpt-5.5-instant","name":"GPT-5.5 Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.3-chat":{"id":"openai/gpt-5.3-chat","name":"GPT-5.3 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16380},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262140,"output":262140},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code-free":{"id":"moonshotai/kimi-k2.7-code-free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"cost":{"input":0.58,"output":3.02,"cache_read":0.1}},"moonshotai/kimi-k3-free":{"id":"moonshotai/kimi-k3-free","name":"Kimi K3 (Free)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2-thinking-turbo":{"id":"moonshotai/kimi-k2-thinking-turbo","name":"Kimi K2 Thinking Turbo","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":1.15,"output":8,"cache_read":0.15}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":0.572,"cache_read":0.058,"cache_write":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.1165,"output":0.2911,"cache_read":0.0233,"tiers":[{"input":0.1747,"output":1.1645,"cache_read":0.0349,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.1456,"output":0.4367,"cache_read":0.0291,"tiers":[{"input":0.2911,"output":0.8734,"cache_read":0.0582,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.98,"output":3.08,"cache_read":0.182}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.375,"output":1.25,"cache_read":0.075}},"z-ai/glm-4.6v-flash-free":{"id":"z-ai/glm-4.6v-flash-free","name":"GLM 4.6V Flash (Free)","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"tiers":[{"input":0,"output":0,"cache_read":0,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.0728,"output":0.4367,"cache_read":0.0146}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.58,"output":2.6,"cache_read":0.14,"tiers":[{"input":0.87,"output":3.18,"cache_read":0.22,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.8781,"output":3.5126,"cache_read":0.1903,"tiers":[{"input":1.1709,"output":4.098,"cache_read":0.2927,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.73,"output":3.19,"cache_read":0.174,"tiers":[{"input":1.02,"output":3.77,"cache_read":0.261,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-image":{"id":"z-ai/glm-image","name":"GLM-Image","description":"GLM-Image is an image generation model adopts a hybrid autoregressive + diffusion decoder architecture. In general image generation quality, GLM‑Image aligns with mainstream latent diffusion approaches, but it shows significant advantages in text-rendering and knowledge‑intensive generation scenarios. It performs especially well in tasks requiring precise semantic understanding and complex information expression, while maintaining strong capabilities in high‑fidelity and fine‑grained detail generation. In addition to text‑to‑image generation, GLM‑Image also supports a rich set of image‑to‑image tasks including image editing, style transfer, identity‑preserving generation, and multi‑subject consistency.","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":10240,"output":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.726,"output":3.1946,"cache_read":0.1743,"tiers":[{"input":1.0165,"output":3.7754,"cache_read":0.2614,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.7-flash-free":{"id":"z-ai/glm-4.7-flash-free","name":"GLM 4.7 Flash (Free)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0}},"z-ai/glm-4.6v-flash":{"id":"z-ai/glm-4.6v-flash","name":"GLM 4.6V FlashX","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.0218,"output":0.2184,"cache_read":0.0044,"tiers":[{"input":0.0437,"output":0.4367,"cache_read":0.0044,"tier":{"type":"context","size":32000}}]}}}},"vancine":{"id":"vancine","env":["VANCINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://vancine.com/v1","name":"Vancine","doc":"https://vancine.com/docs","models":{"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.24,"output":0.96,"cache_read":0.0048}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.67,"output":2,"cache_read":0.034}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.4,"output":12,"cache_read":0.24}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.4,"cache_read":0.024}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.013}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6,"output":4.8,"cache_read":0.2}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.24,"output":0.96,"cache_read":0.048}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.12,"output":3.52,"cache_read":0.208}}}},"minimax-cn":{"id":"minimax-cn","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.cn/anthropic/v1","name":"MiniMax (minimax.cn)","doc":"https://platform.minimaxi.com/docs/guides/quickstart","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}}}},"cortecs":{"id":"cortecs","env":["CORTECS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cortecs.ai/v1","name":"Cortecs","doc":"https://api.cortecs.ai/v1/models","models":{"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Ministral 3 14B is a frontier-level 14B multimodal model optimized for local deployment, delivering state-of-the-art text and vision reasoning with a 256K context window and strong agentic capabilities.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.24,"output":0.24,"cache_read":0.022}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.06,"output":0.439,"cache_read":0.019}},"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"qwen3guard-gen-0.6b","description":"Qwen3Guard-Gen-0.6B is a lightweight multilingual safety moderation model that classifies prompts and responses into safe, controversial, or unsafe categories.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":81920},"cost":{"input":0.111,"output":0.557}},"nova-2-lite":{"id":"nova-2-lite","name":"Nova 2 Lite","description":"Nova 2 Lite is an advanced multimodal reasoning model that combines efficiency and performance, delivering reliable AI for agentic workflows and enterprise applications.","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.373,"output":3.144}},"mistral-small-2503":{"id":"mistral-small-2503","name":"mistral-small-2503","description":"Combines advanced text and vision capabilities with 24 billion parameters, supporting multilingual tasks and long contexts up to 131k tokens, making it versatile for various applications without sacrificing performance.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.111,"output":0.334}},"mistral-7b-instruct-v0.2":{"id":"mistral-7b-instruct-v0.2","name":"mistral-7b-instruct-v0.2","description":"Mistral 7B Instruct is a compact, 7B parameter model optimized for fast and efficient text and code generation with a 32K token context window.","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.159,"output":0.219}},"codestral-2508":{"id":"codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.368,"output":1.103,"cache_read":0.037}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Optimized for dialogue, this LLM by Meta outperforms other open-source chat models in benchmarks while prioritizing helpfulness and safety.","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.167,"output":0.167}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":2,"output":3.999,"cache_read":0.5}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.111,"output":0.434,"cache_read":0.056}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.055,"output":0.174,"cache_read":0.009}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.201,"output":0.5,"cache_read":0.05}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.359,"output":1.435}},"claude-4-5-sonnet":{"id":"claude-4-5-sonnet","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.989,"output":14.945,"cache_read":0.326,"cache_write":4.078}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.111,"output":0.167}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.478,"output":2.392,"cache_read":0.045}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":400000,"output":196000},"cost":{"input":0.349,"output":1.405}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":32.998,"cache_read":0.55,"cache_write":6.879}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.4,"cache_read":0.04}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.498,"cache_read":0.55,"cache_write":6.874}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.668,"output":2.674}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.516,"output":2.869,"cache_read":0.115}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1000000},"cost":{"input":1.1,"output":2.99,"cache_read":0.18}},"claude-opus4-5":{"id":"claude-opus4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.313,"output":26.568,"cache_read":0.531,"cache_write":6.645}},"claude-opus4-6":{"id":"claude-opus4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.313,"output":26.561,"cache_read":0.531,"cache_write":6.645}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.296,"output":1.186,"cache_read":0.075}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.395,"output":1.977,"cache_read":0.099}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.706,"output":3.208,"cache_read":0.18}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.179,"output":0.697}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Apertus 70B is an open, multilingual language model designed for research, long-context reasoning, and sovereignty-focused AI systems.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":1.393,"output":2.228}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.434,"output":1.704,"cache_read":0.134}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.898,"output":15.453,"cache_read":0.242}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.045,"output":0.167}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.272,"output":1.631,"cache_read":0.025,"cache_write":0.082}},"mistral-large-2402":{"id":"mistral-large-2402","name":"mistral-large-2402","description":"Mistral Large (24.02) is Mistral AI’s most advanced language model, built for complex multilingual reasoning, code generation, and deep text understanding.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":4.284,"output":12.952}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"qwen2.5-vl-72b-instruct","description":"Qwen2.5-VL is a powerful vision-language model with advanced capabilities in visual understanding, long video reasoning, and structured output generation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":1.014,"output":1.014}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.5,"output":1.499,"cache_read":0.13}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.167,"output":0.891}},"claude-opus4-7":{"id":"claude-opus4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.067,"output":0.245,"cache_read":0.014}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.668,"output":4.01}},"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.179,"output":0.697}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.613,"output":1.838,"cache_read":0.061}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.649,"output":9.899,"cache_read":0.165,"cache_write":1}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.446,"output":3.008}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":2.659,"output":10.635,"cache_read":1.33}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.219,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"ministral-3b-2512","description":"Ministral 3 3B is a compact, efficient multimodal model with strong language, vision capabilities, and ideal for custom fine-tuning.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.123,"output":0.123,"cache_read":0.012}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":14.999}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Gemma 3 is a family of lightweight, multimodal models from Google, supporting text and image inputs, multilingual capabilities, and a 131K context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":110000},"cost":{"input":0.099,"output":0.299}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.652,"output":2.57,"cache_read":0.163}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.296,"output":0.495,"cache_read":0.075}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":32768},"cost":{"input":0.167,"output":0.557}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"mistral-nemo-instruct-2407","description":"A 12B parameter, instruct-tuned language model by Mistral AI and NVIDIA, designed for advanced instruction following, multi-turn conversations, and generating text and code across multiple languages.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-07","last_updated":"2024-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.145,"output":0.145,"cache_read":0.014}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.35,"cache_read":0.018}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.159,"output":0.638,"cache_read":0.081}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.749,"cache_read":0.033}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.625}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.192,"output":8.769,"cache_read":0.546}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":3.46,"cache_read":0.124}},"claude-opus4-8":{"id":"claude-opus4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"pixtral-large-2502":{"id":"pixtral-large-2502","name":"Pixtral Large (25.02)","description":"Pixtral Large (25.02) is a 124B open-weight multimodal model built on Mistral Large 2, offering advanced image understanding and strong performance across text and code tasks.","family":"pixtral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.993,"output":5.978}},"nova-micro-v1":{"id":"nova-micro-v1","name":"nova-micro-v1","description":"Nova Micro is a multilingual text-to-text foundation model with strong reasoning capabilities and broad language coverage across 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.04,"output":0.159}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"qwen3-30b-a3b-instruct-2507","description":"Qwen3-30B-A3B-Instruct-2507 is an advanced Mixture-of-Experts model optimized for reasoning, coding, and multilingual instruction following.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.099,"output":0.299}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"mistral-small-3.2-24b-instruct-2506","description":"Mistral-Small-3.2-24B-Instruct-2506 is a 24B parameter instruction-tuned model with enhanced long-context support (128k) and state-of-the-art vision understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.1,"output":0.312}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"mistral-7b-instruct-v0.3","description":"Mistral-7B-Instruct-v0.3 model is a fine-tuned version of the Mistral 7B base model, optimized for instruction-following tasks. Released in 2023, it is intended for demonstration purposes and does not include built-in guardrails or moderation features.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":127000},"cost":{"input":0.111,"output":0.111}},"nova-pro-v1":{"id":"nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.918,"output":3.671}},"mixtral-8x7B-instruct-v0.1":{"id":"mixtral-8x7B-instruct-v0.1","name":"Mixtral 8x7B Instruct v0.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.488,"output":0.758}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.223,"output":0.39}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.996,"output":4.982,"cache_read":0.099,"cache_write":1.186}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.988,"output":3.164,"cache_read":0.247}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"qwen3guard-gen-8b","description":"Qwen3Guard-Gen-8B is a large-scale multilingual safety moderation model designed for high-accuracy prompt and response classification.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"nvidia-nemotron-3-nano-30b-a3b","description":"Nemotron-Nano-3-30B-A3B is a compact Mixture-of-Experts model optimized for efficient reasoning, chat, and coding, with strong multilingual support and long-context RAG and agent workflows.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-12","last_updated":"2026-01-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.06,"output":0.24}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":2.768,"cache_read":0.124}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.384,"output":4.348,"cache_read":0.346}},"hermes-4-405b":{"id":"hermes-4-405b","name":"hermes-4-405b","description":"Hermes 4 405B is a frontier hybrid-mode reasoning model built on Llama 3.1, optimized for advanced logic, math, coding, and structured output generation.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.996,"output":2.989}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.069,"output":0.455,"cache_read":0.018}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.156,"output":0.625,"cache_read":0.016}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.082,"cache_write":0.084}},"claude-4-6-sonnet":{"id":"claude-4-6-sonnet","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.196,"output":15.94,"cache_read":0.32,"cache_write":3.999}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.73,"output":3.46,"cache_read":0.432}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"ministral-8b-2512","description":"Ministral 3 8B is a balanced, efficient multimodal model offering strong text and vision capabilities, optimized for edge and local deployment.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.179,"output":0.179,"cache_read":0.017}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.279,"output":2.192,"cache_read":0.056}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.089,"output":0.446,"cache_read":0.01}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.495,"output":9.964,"cache_read":0.242,"cache_write":0.434}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.199,"cache_read":0.219,"cache_write":2.749}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.114,"output":3.899,"cache_read":0.279}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"mistral-medium-3.5","description":"Mistral Medium 3.5 is a frontier multimodal 128B model combining reasoning, coding, and instruction-following with strong agentic performance and efficient deployment.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1.532,"output":7.843,"cache_read":0.154}},"minicpm-v-4.5":{"id":"minicpm-v-4.5","name":"minicpm-v-4.5","description":"MiniCPM-V 4.5 is a compact, high-performance vision-language model excelling in video understanding, OCR, and multimodal reasoning with efficient deployment.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.651,"output":1.097}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"pixtral-12b-2409","description":"Pixtral 2409 12B is a state-of-the-art multimodal model with 12B parameters and a 400M vision encoder, natively trained on interleaved text and image data. It excels in tasks spanning vision-language reasoning, instruction following, and pure text understanding, making it highly effective for real-world multimodal applications.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-11-09","last_updated":"2024-11-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.223,"output":0.223}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.299,"output":2.491,"cache_read":0.029,"cache_write":0.097}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65000},"cost":{"input":2.898,"output":14.493,"cache_read":0.29,"cache_write":3.624}},"voxtral-small-2507":{"id":"voxtral-small-2507","name":"voxtral-small-2507","description":"Voxtral Small is a multimodal model with audio input, combining advanced speech capabilities with strong text performance for transcription, translation, and audio understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.123,"output":0.368,"cache_read":0.012}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.219,"cache_write":2.749}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"qwen3-vl-235b-a22b","description":"Qwen3 VL 235B A22B is a 235B-parameter MoE vision-language flagship model (≈22B active) designed for frontier-level multimodal understanding across text, images, documents, and long videos.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.617,"output":3.119,"cache_read":0.052}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.724,"output":0.724}},"nova-lite-v1":{"id":"nova-lite-v1","name":"nova-lite-v1","description":"Nova Lite is a fast, low-cost multimodal foundation model capable of reasoning over text, images, and video in 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.069,"output":0.275}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":203000},"cost":{"input":0.08,"output":0.478}},"nemotron-nano-v2-12b":{"id":"nemotron-nano-v2-12b","name":"nemotron-nano-v2-12b","description":"NVIDIA Nemotron Nano v2 12B is a 12-billion-parameter multimodal reasoning model designed for advanced video understanding, document intelligence, and visual reasoning, built with a hybrid Transformer-Mamba architecture for high efficiency and low latency.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.24,"output":0.707}}}},"wallaby":{"id":"wallaby","env":["WALLABY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.wallabytoken.com/v1","name":"Wallaby","doc":"https://wallabytoken.com/docs","models":{"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.7,"output":13.5,"cache_read":0.27}}}},"ainetcafe":{"id":"ainetcafe","env":["AINETCAFE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://microquickjs.com/v1","name":"ainetcafe","doc":"https://ainetcafe.com/k3/guides/","models":{"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":2.1,"output":10.5,"cache_read":0.3}}}},"hpc-ai":{"id":"hpc-ai","env":["HPC_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.hpc-ai.com/inference/v1","name":"HPC-AI","doc":"https://www.hpc-ai.com/doc/docs/quickstart/","models":{"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":195000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":202000},"cost":{"input":0.615,"output":2.46,"cache_read":0.133}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1002000,"output":128000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3,"cache_read":0.1}}}},"tencent-coding-plan":{"id":"tencent-coding-plan","env":["TENCENT_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/coding/v3","name":"Tencent Coding Plan (China)","doc":"https://cloud.tencent.com/document/product/1772/128947","models":{"hunyuan-2.0-thinking":{"id":"hunyuan-2.0-thinking","name":"Tencent HY 2.0 Think","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-t1":{"id":"hunyuan-t1","name":"Hunyuan-T1","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-turbos":{"id":"hunyuan-turbos","name":"Hunyuan-TurboS","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"tc-code-latest":{"id":"tc-code-latest","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-2.0-instruct":{"id":"hunyuan-2.0-instruct","name":"Tencent HY 2.0 Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"v0":{"id":"v0","env":["V0_API_KEY"],"npm":"@ai-sdk/vercel","name":"v0","doc":"https://sdk.vercel.ai/providers/ai-sdk-providers/vercel","models":{"v0-1.5-lg":{"id":"v0-1.5-lg","name":"v0-1.5-lg","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":32000},"cost":{"input":15,"output":75}},"v0-1.5-md":{"id":"v0-1.5-md","name":"v0-1.5-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}},"v0-1.0-md":{"id":"v0-1.0-md","name":"v0-1.0-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}}}},"nan":{"id":"nan","env":["NAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nan.builders/v1","name":"NaN","doc":"https://nan.builders/docs/models","models":{"glm5.3":{"id":"glm5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"qwen3.6":{"id":"qwen3.6","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gemma4":{"id":"gemma4","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"glm5.3-flash":{"id":"glm5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}}}},"ai21":{"id":"ai21","env":["AI21_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ai21.com/studio/v1","name":"AI21 Labs","doc":"https://docs.ai21.com/docs/jamba-foundation-models","models":{"jamba-large":{"id":"jamba-large","name":"Jamba Large","description":"AI21's hybrid SSM-Transformer long-context model for enterprise agents and grounded generation","family":"jamba","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-22","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":2,"output":8}},"jamba-mini":{"id":"jamba-mini","name":"Jamba Mini","description":"AI21's efficient, lightweight hybrid SSM-Transformer model for a wide range of tasks","family":"jamba","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-22","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.2,"output":0.4}}}},"perplexity":{"id":"perplexity","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/perplexity","name":"Perplexity","doc":"https://docs.perplexity.ai","models":{"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"alibaba-token-plan-cn":{"id":"alibaba-token-plan-cn","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan (China)","doc":"https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}}}},"oci":{"id":"oci","env":["OCI_GENAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.generativeai.us-chicago-1.oci.oraclecloud.com/openai/v1","name":"OCI Generative AI","doc":"https://docs.oracle.com/en-us/iaas/Content/generative-ai/pretrained-models.htm","models":{"meta.llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta.llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":16384},"cost":{"input":0.72,"output":0.72}},"meta.llama-4-scout-17b-16e-instruct":{"id":"meta.llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":192000,"output":16384},"cost":{"input":0.72,"output":0.72}},"meta.llama-3.3-70b-instruct":{"id":"meta.llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"xai.grok-4.6":{"id":"xai.grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai.grok-4.20-reasoning":{"id":"xai.grok-4.20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"openai.gpt-oss-20b":{"id":"openai.gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"xai.grok-4.3":{"id":"xai.grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"openai.gpt-oss-120b":{"id":"openai.gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"xai.grok-4.20-non-reasoning":{"id":"xai.grok-4.20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}}}},"drun":{"id":"drun","env":["DRUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://chat.d.run/v1","name":"D.Run (China)","doc":"https://www.d.run","models":{"public/deepseek-v3":{"id":"public/deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.28,"output":1.1}},"public/minimax-m25":{"id":"public/minimax-m25","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"temperature":true,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.29,"output":1.16}},"public/deepseek-r1":{"id":"public/deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.55,"output":2.2}}}},"google-vertex-anthropic":{"id":"google-vertex-anthropic","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex/anthropic","name":"Vertex (Anthropic)","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude","models":{"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}}}},"anyapi":{"id":"anyapi","env":["ANYAPI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.anyapi.ai/v1","name":"AnyAPI","doc":"https://docs.anyapi.ai","models":{"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated"},"mistralai/mistral-large-2512":{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}}}},"opencode-go":{"id":"opencode-go","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/go/v1","name":"OpenCode Go","doc":"https://opencode.ai/docs/go","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.7-max","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"ox-alpha-free":{"id":"ox-alpha-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"omen-alpha":{"id":"omen-alpha","name":"Omen Alpha","description":"oH man anothEr aLPha ModEl","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":128000},"status":"deprecated","cost":{"input":0.2,"output":0.66,"cache_read":0.04}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo V2 Pro","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"status":"deprecated","cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-09-22","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"status":"deprecated","cost":{"input":1,"output":3.2,"cache_read":0.2}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo V2 Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-omni","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2,"cache_read":0.08}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter multimodal flagship for coding, professional work, and long-horizon agentic workflows","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.6,"output":3,"cache_read":0.1}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.7-plus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro (New)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo-v2.5-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Legacy model retained for compatibility with older integrations","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}}}},"tencent-token-plan":{"id":"tencent-token-plan","env":["TENCENT_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/plan/v3","name":"Tencent Token Plan","doc":"https://cloud.tencent.com/document/product/1823/130060","models":{"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}}}},"gitlab":{"id":"gitlab","env":["GITLAB_TOKEN"],"npm":"gitlab-ai-provider","name":"GitLab Duo","doc":"https://docs.gitlab.com/user/duo_agent_platform/","models":{"duo-chat-gpt-5-6-luna":{"id":"duo-chat-gpt-5-6-luna","name":"Agentic Chat (GPT-5.6 Luna)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-5":{"id":"duo-chat-opus-5","name":"Agentic Chat (Claude Opus 5)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-8":{"id":"duo-chat-opus-4-8","name":"Agentic Chat (Claude Opus 4.8)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-1":{"id":"duo-chat-gpt-5-1","name":"Agentic Chat (GPT-5.1)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-2":{"id":"duo-chat-gpt-5-2","name":"Agentic Chat (GPT-5.2)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-4-nano":{"id":"duo-chat-gpt-5-4-nano","name":"Agentic Chat (GPT-5.4 Nano)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-haiku-4-5":{"id":"duo-chat-haiku-4-5","name":"Agentic Chat (Claude Haiku 4.5)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-6":{"id":"duo-chat-opus-4-6","name":"Agentic Chat (Claude Opus 4.6)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-6-terra":{"id":"duo-chat-gpt-5-6-terra","name":"Agentic Chat (GPT-5.6 Terra)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-sonnet-5":{"id":"duo-chat-sonnet-5","name":"Agentic Chat (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-5":{"id":"duo-chat-opus-4-5","name":"Agentic Chat (Claude Opus 4.5)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-6-astra":{"id":"duo-chat-gpt-6-astra","name":"Agentic Chat (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-3-codex":{"id":"duo-chat-gpt-5-3-codex","name":"Agentic Chat (GPT-5.3 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-fable-5-1":{"id":"duo-chat-fable-5-1","name":"Agentic Chat (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-4-mini":{"id":"duo-chat-gpt-5-4-mini","name":"Agentic Chat (GPT-5.4 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-sonnet-4-6":{"id":"duo-chat-sonnet-4-6","name":"Agentic Chat (Claude Sonnet 4.6)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-5":{"id":"duo-chat-gpt-5-5","name":"Agentic Chat (GPT-5.5)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-fable-5":{"id":"duo-chat-fable-5","name":"Agentic Chat (Claude Fable 5)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-7":{"id":"duo-chat-opus-4-7","name":"Agentic Chat (Claude Opus 4.7)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-4":{"id":"duo-chat-gpt-5-4","name":"Agentic Chat (GPT-5.4)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-6-sol":{"id":"duo-chat-gpt-5-6-sol","name":"Agentic Chat (GPT-5.6 Sol)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-2-codex":{"id":"duo-chat-gpt-5-2-codex","name":"Agentic Chat (GPT-5.2 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-codex":{"id":"duo-chat-gpt-5-codex","name":"Agentic Chat (GPT-5 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-mini":{"id":"duo-chat-gpt-5-mini","name":"Agentic Chat (GPT-5 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-sonnet-4-5":{"id":"duo-chat-sonnet-4-5","name":"Agentic Chat (Claude Sonnet 4.5)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"vispark":{"id":"vispark","env":["VISPARK_LAB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lab.vispark.in/v1","name":"Vispark","doc":"https://lab.vispark.in/#vision","models":{"vispark/vision-large":{"id":"vispark/vision-large","name":"Vision Large","description":"Most capable Vision model for complex reasoning, detailed media analysis, and structured output over a 1M-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":7.37,"output":22.11}},"vispark/vision-medium":{"id":"vispark/vision-medium","name":"Vision Medium","description":"Balanced multimodal model pairing a 1M-token context window with deeper reasoning for analysis, content creation, and tool use across text, image, audio, video, and PDF inputs.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":4.21,"output":12.63}},"vispark/vision-small":{"id":"vispark/vision-small","name":"Vision Small","description":"Fast, low-cost multimodal model for understanding text, images, audio, video, and PDFs, with tool calling and a 1M-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.05,"output":3.16}}}},"neosmith":{"id":"neosmith","env":["NEOSMITH_API_KEY"],"npm":"@ai-sdk/openai","api":"https://router.neosmith.ai/v1","name":"NeoSmith","doc":"https://neosmith.ai/docs","models":{"neosmith.intelligent-maestro":{"id":"neosmith.intelligent-maestro","name":"NeoSmith Maestro","description":"Highest-accuracy coding tier. Hard, self-contained problems run NeoSmith's premium multi-model solver; everything else gets the strongest intelligence tier.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.4,"output":12,"cache_read":0.35,"cache_write":0}},"neosmith.intelligent-basic":{"id":"neosmith.intelligent-basic","name":"NeoSmith Basic","description":"Cost-capped tier. Intelligent routing with a Claude Sonnet ceiling — Opus is never invoked.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.17,"output":4.37,"cache_read":0.22,"cache_write":0}},"neosmith.intelligent-pro":{"id":"neosmith.intelligent-pro","name":"NeoSmith Pro","description":"Default production tier. Intelligent NeoSmith routing with a Claude Opus ceiling on escalation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.81,"output":8.39,"cache_read":0.3,"cache_write":0}},"neosmith.neolite":{"id":"neosmith.neolite","name":"NeoSmith NeoLite","description":"Sealed single-model budget tier. 512K context, text and images, tool use, and no escalation of any kind.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":64000},"cost":{"input":0.6,"output":2.4,"cache_read":0.08,"cache_write":0}}}},"tinfoil":{"id":"tinfoil","env":["TINFOIL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.tinfoil.sh/v1","name":"Tinfoil","doc":"https://docs.tinfoil.sh","models":{"nomic-embed-text":{"id":"nomic-embed-text","name":"Nomic Embed Text v1.5","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2024-02","last_updated":"2024-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":768},"cost":{"input":0.05,"output":0}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":0.65,"output":1.45,"cache_read":0.13}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":1}},"glm-5-3":{"id":"glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.8,"output":5.75,"cache_read":0.45}},"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"gpt-oss-safeguard-120b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":4,"output":20,"cache_read":0.8}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":1.25,"cache_read":0.1}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"llama3-3-70b":{"id":"llama3-3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":1.75,"output":2.75}}}},"edenai":{"id":"edenai","env":["EDENAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.edenai.run/v3","name":"Eden AI","doc":"https://docs.edenai.co","models":{"qwen/deepseek-v4-pro-0813":{"id":"qwen/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Alibaba)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.066}},"qwen/deepseek-v4-flash-0731":{"id":"qwen/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Alibaba)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.022}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen-vl-max":{"id":"qwen/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.16}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4,"cache_read":0.32}},"qwen/qwen-vl-plus":{"id":"qwen/qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63,"cache_read":0.042}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"qwen/qwen3-max@eu":{"id":"qwen/qwen3-max@eu","name":"Qwen3 Max (EU)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwq-plus":{"id":"qwen/qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen/deepseek-v4.1-flash":{"id":"qwen/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Alibaba)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-coder-next@eu":{"id":"qwen/qwen3-coder-next@eu","name":"Qwen3 Coder Next (EU)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.23,"output":0.92}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5}},"groq/openai/gpt-oss-20b":{"id":"groq/openai/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/openai/gpt-oss-safeguard-20b":{"id":"groq/openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B (Groq)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/openai/gpt-oss-120b":{"id":"groq/openai/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"scaleway/deepseek-v4-flash-0731":{"id":"scaleway/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Scaleway)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":384000},"cost":{"input":0.4596,"output":0.9192}},"scaleway/gemma-3-27b-it":{"id":"scaleway/gemma-3-27b-it","name":"Gemma 3 27B IT (Scaleway)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40000,"output":131072},"cost":{"input":0.287125,"output":0.57425}},"scaleway/gpt-oss-120b":{"id":"scaleway/gpt-oss-120b","name":"GPT OSS 120B (Scaleway)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.17235,"output":0.6894}},"scaleway/llama-3.3-70b-instruct":{"id":"scaleway/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct (Scaleway)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.0341,"output":1.0341}},"nebius/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"nebius/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Nebius)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"nebius/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"nebius/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Nebius)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"nebius/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"nebius/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Nebius)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":979000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"nebius/nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nebius/nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Nebius)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":1,"output":3,"cache_read":1}},"nebius/nvidia/nemotron-3-super-120b-a12b":{"id":"nebius/nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B (Nebius)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.9,"cache_read":0.3}},"nebius/google/gemma-3-27b-it":{"id":"nebius/google/gemma-3-27b-it","name":"Gemma 3 27B IT (Nebius)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":110000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.1}},"nebius/openai/gpt-oss-120b":{"id":"nebius/openai/gpt-oss-120b","name":"GPT OSS 120B (Nebius)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5-Coder-32B-Instruct (Cloudflare)","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.66,"output":1}},"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Cloudflare)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Cloudflare)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"cloudflare/@cf/zai-org/glm-4.7-flash":{"id":"cloudflare/@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash (Cloudflare)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma-SEA-LION-v4-27B-IT (Cloudflare)","description":"Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"cloudflare/@cf/meta/llama-guard-3-8b":{"id":"cloudflare/@cf/meta/llama-guard-3-8b","name":"Llama-Guard-3-8B (Cloudflare)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.484,"output":0.03}},"cloudflare/@cf/openai/gpt-oss-20b":{"id":"cloudflare/@cf/openai/gpt-oss-20b","name":"GPT OSS 20B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.3}},"cloudflare/@cf/openai/gpt-oss-120b":{"id":"cloudflare/@cf/openai/gpt-oss-120b","name":"GPT OSS 120B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.35,"output":0.75}},"minimax/MiniMax-M2":{"id":"minimax/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"minimax/MiniMax-M2.1":{"id":"minimax/MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M2.5":{"id":"minimax/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M3":{"id":"minimax/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/MiniMax-M2.7":{"id":"minimax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"ovhcloud/gpt-oss-20b":{"id":"ovhcloud/gpt-oss-20b","name":"GPT OSS 20B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.18}},"ovhcloud/gpt-oss-120b":{"id":"ovhcloud/gpt-oss-120b","name":"GPT OSS 120B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.47}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest (Claude Opus 5)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["audio","image","text","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"tensorx/deepseek/deepseek-v4-pro-0813":{"id":"tensorx/deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (TensorX)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":2,"output":4,"cache_read":0.5}},"tensorx/deepseek/deepseek-v4-flash-0731":{"id":"tensorx/deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (TensorX)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.0625}},"tensorx/deepseek/deepseek-v4.1-flash":{"id":"tensorx/deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (TensorX)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.5,"output":1.5,"cache_read":0.125}},"tensorx/moonshotai/kimi-k2.5":{"id":"tensorx/moonshotai/kimi-k2.5","name":"Kimi K2.5 (TensorX)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125}},"infomaniak/mistralai/Ministral-3-14B-Instruct-2512":{"id":"infomaniak/mistralai/Ministral-3-14B-Instruct-2512","name":"Ministral 3 14B (Infomaniak)","description":"Open vision-language model for efficient local deployment, instruction following, and tool use","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":262144},"cost":{"input":0.3447,"output":0.4596}},"flexai/Step-3.7-Flash":{"id":"flexai/Step-3.7-Flash","name":"Step 3.7 Flash (FlexAI)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15}},"flexai/gpt-oss-20b":{"id":"flexai/gpt-oss-20b","name":"GPT OSS 20B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.13}},"flexai/DeepSeek-V4-Flash-0731":{"id":"flexai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (FlexAI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.065,"output":0.18}},"flexai/Muse-Glimmer-30B":{"id":"flexai/Muse-Glimmer-30B","name":"Muse Glimmer 30B (FlexAI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.1}},"flexai/gpt-oss-120b":{"id":"flexai/gpt-oss-120b","name":"GPT OSS 120B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.037,"output":0.17}},"databricks/databricks-gpt-oss-20b@eu":{"id":"databricks/databricks-gpt-oss-20b@eu","name":"GPT OSS 20B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"databricks/databricks-deepseek-v4-pro-0813":{"id":"databricks/databricks-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Databricks)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.31999,"output":3.95997,"cache_read":0.13202,"cache_write":1.31999}},"databricks/databricks-gpt-oss-120b@eu":{"id":"databricks/databricks-gpt-oss-120b@eu","name":"GPT OSS 120B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"databricks/databricks-gpt-oss-20b":{"id":"databricks/databricks-gpt-oss-20b","name":"GPT OSS 20B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"databricks/databricks-deepseek-v4-flash-0731":{"id":"databricks/databricks-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Databricks)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028,"cache_write":0.14}},"databricks/databricks-inkling":{"id":"databricks/databricks-inkling","name":"Inkling (Databricks)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1048576},"cost":{"input":1.00002,"output":4.04999,"cache_read":0.17003,"cache_write":1.00002}},"databricks/databricks-gpt-oss-120b":{"id":"databricks/databricks-gpt-oss-120b","name":"GPT OSS 120B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"deepinfra/nemotron-3-ultra-550b-a55b":{"id":"deepinfra/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Deep Infra)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/ByteDance/Seed-2.0-mini":{"id":"deepinfra/ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini (Deep Infra)","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"deepinfra/ByteDance/Seed-2.0-code":{"id":"deepinfra/ByteDance/Seed-2.0-code","name":"Seed 2.0 Code (Deep Infra)","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"deepinfra/stepfun-ai/Step-3.7-Flash":{"id":"deepinfra/stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash (Deep Infra)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepinfra/stepfun-ai/Step-3.5-Flash":{"id":"deepinfra/stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash (Deep Infra)","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"deepinfra/deepseek-ai/DeepSeek-V3":{"id":"deepinfra/deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3 (Deep Infra)","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepinfra/deepseek-ai/DeepSeek-V3-0324":{"id":"deepinfra/deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324 (Deep Infra)","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Deep Infra)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Deep Infra)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Deep Infra)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/deepseek-ai/DeepSeek-R1":{"id":"deepinfra/deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1 (Deep Infra)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.4}},"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct":{"id":"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct","name":"Llama 3.1 Nemotron 70B Instruct (Deep Infra)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.6,"output":0.6}},"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B":{"id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B (Deep Infra)","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"deepinfra/meta-models/Muse-Glimmer-30B":{"id":"deepinfra/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Deep Infra)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"deepinfra/google/gemma-3-4b-it":{"id":"deepinfra/google/gemma-3-4b-it","name":"Gemma 3 4B IT (Deep Infra)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"deepinfra/google/gemma-3-27b-it":{"id":"deepinfra/google/gemma-3-27b-it","name":"Gemma 3 27B IT (Deep Infra)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"deepinfra/google/gemma-3-12b-it":{"id":"deepinfra/google/gemma-3-12b-it","name":"Gemma 3 12B IT (Deep Infra)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"deepinfra/zai-org/GLM-4.7-Flash":{"id":"deepinfra/zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash (Deep Infra)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"deepinfra/thinkingmachines/Inkling-Small":{"id":"deepinfra/thinkingmachines/Inkling-Small","name":"Inkling Small (Deep Infra)","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"deepinfra/thinkingmachines/Inkling":{"id":"deepinfra/thinkingmachines/Inkling","name":"Inkling (Deep Infra)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"deepinfra/meta-llama/Llama-Guard-3-8B":{"id":"deepinfra/meta-llama/Llama-Guard-3-8B","name":"Llama-Guard-3-8B (Deep Infra)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.055,"output":0.055}},"deepinfra/meta-llama/Llama-3.3-70B-Instruct":{"id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (Deep Infra)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.1,"output":0.32}},"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct":{"id":"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct","name":"Llama-3.2-11B-Vision-Instruct (Deep Infra)","description":"Open multimodal Llama model for image understanding, captioning, and visual QA","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.345,"output":0.345}},"deepinfra/openai/gpt-oss-20b":{"id":"deepinfra/openai/gpt-oss-20b","name":"GPT OSS 20B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.14}},"deepinfra/openai/gpt-oss-120b":{"id":"deepinfra/openai/gpt-oss-120b","name":"GPT OSS 120B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.037,"output":0.17}},"deepinfra/moonshotai/Kimi-K2.5":{"id":"deepinfra/moonshotai/Kimi-K2.5","name":"Kimi K2.5 (Deep Infra)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"deepinfra/tencent/Hy3":{"id":"deepinfra/tencent/Hy3","name":"Hy3 (Deep Infra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5.1-codex-max":{"id":"azure/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"fireworks_ai/gpt-oss-120b":{"id":"fireworks_ai/gpt-oss-120b","name":"GPT OSS 120B (Fireworks AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.014}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Fireworks AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Fireworks AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b":{"id":"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B (Fireworks AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"fireworks_ai/accounts/fireworks/models/inkling":{"id":"fireworks_ai/accounts/fireworks/models/inkling","name":"Inkling (Fireworks AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"amazon/moonshotai.kimi-k2.5":{"id":"amazon/moonshotai.kimi-k2.5","name":"Kimi K2.5 (Amazon Bedrock)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}},"amazon/amazon.nova-micro-v1:0@us":{"id":"amazon/amazon.nova-micro-v1:0@us","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"amazon/mistral.pixtral-large-2502-v1:0":{"id":"amazon/mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (Amazon Bedrock)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/amazon.nova-lite-v1:0@us":{"id":"amazon/amazon.nova-lite-v1:0@us","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"amazon/zai.glm-4.7-flash@us":{"id":"amazon/zai.glm-4.7-flash@us","name":"GLM-4.7-Flash (Amazon Bedrock, US)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"amazon/google.gemma-3-12b-it@us":{"id":"amazon/google.gemma-3-12b-it@us","name":"Gemma 3 12B IT (Amazon Bedrock, US)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.09,"output":0.29}},"amazon/mistral.voxtral-mini-3b-2507":{"id":"amazon/mistral.voxtral-mini-3b-2507","name":"Voxtral Mini 3B 2507 (Amazon Bedrock)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.04,"output":0.04}},"amazon/amazon.nova-pro-v1:0@us":{"id":"amazon/amazon.nova-pro-v1:0@us","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"amazon/google.gemma-3-12b-it":{"id":"amazon/google.gemma-3-12b-it","name":"Gemma 3 12B IT (Amazon Bedrock)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.09,"output":0.29}},"amazon/openai.gpt-oss-safeguard-20b":{"id":"amazon/openai.gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B (Amazon Bedrock)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.07,"output":0.2}},"amazon/amazon.nova-lite-v1:0":{"id":"amazon/amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"amazon/amazon.nova-pro-v1:0":{"id":"amazon/amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"amazon/openai.gpt-oss-safeguard-20b@us":{"id":"amazon/openai.gpt-oss-safeguard-20b@us","name":"GPT OSS Safeguard 20B (Amazon Bedrock, US)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.07,"output":0.2}},"amazon/moonshot.kimi-k2-thinking":{"id":"amazon/moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking (Amazon Bedrock)","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":0.6,"output":2.5}},"amazon/google.gemma-3-27b-it":{"id":"amazon/google.gemma-3-27b-it","name":"Gemma 3 27B IT (Amazon Bedrock)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.23,"output":0.38}},"amazon/amazon.nova-micro-v1:0":{"id":"amazon/amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"amazon/mistral.voxtral-mini-3b-2507@us":{"id":"amazon/mistral.voxtral-mini-3b-2507@us","name":"Voxtral Mini 3B 2507 (Amazon Bedrock, US)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.04,"output":0.04}},"amazon/mistral.pixtral-large-2502-v1:0@us":{"id":"amazon/mistral.pixtral-large-2502-v1:0@us","name":"Pixtral Large (25.02) (Amazon Bedrock, US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/google.gemma-3-27b-it@us":{"id":"amazon/google.gemma-3-27b-it@us","name":"Gemma 3 27B IT (Amazon Bedrock, US)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.23,"output":0.38}},"amazon/zai.glm-4.7-flash":{"id":"amazon/zai.glm-4.7-flash","name":"GLM-4.7-Flash (Amazon Bedrock)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"amazon/google.gemma-3-4b-it":{"id":"amazon/google.gemma-3-4b-it","name":"Gemma 3 4B IT (Amazon Bedrock)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.04,"output":0.08}},"amazon/mistral.voxtral-small-24b-2507":{"id":"amazon/mistral.voxtral-small-24b-2507","name":"Voxtral Small 24B 2507 (Amazon Bedrock)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"amazon/mistral.voxtral-small-24b-2507@us":{"id":"amazon/mistral.voxtral-small-24b-2507@us","name":"Voxtral Small 24B 2507 (Amazon Bedrock, US)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"amazon/google.gemma-3-4b-it@us":{"id":"amazon/google.gemma-3-4b-it@us","name":"Gemma 3 4B IT (Amazon Bedrock, US)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.04,"output":0.08}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-mini-latest":{"id":"openai/gpt-mini-latest","name":"GPT Mini Latest (GPT-5.4 mini)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-pro-latest":{"id":"openai/gpt-pro-latest","name":"GPT Pro Latest (GPT-5.5 Pro)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":288000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":132000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-latest":{"id":"xai/grok-latest","name":"Grok Latest (Grok 4.6)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Together AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"together_ai/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"together_ai/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Together AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Together AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together_ai/meta-models/Muse-Glimmer-30B":{"id":"together_ai/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Together AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"together_ai/thinkingmachines/Inkling":{"id":"together_ai/thinkingmachines/Inkling","name":"Inkling (Together AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"together_ai/openai/gpt-oss-120b":{"id":"together_ai/openai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-medium-2604":{"id":"mistral/mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistral/mistral-small-2603":{"id":"mistral/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"vertex/gemini-pro-latest":{"id":"vertex/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview, Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.1-flash-lite-image":{"id":"vertex/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite (Vertex AI)","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"vertex/gemini-3.7-flash@eu":{"id":"vertex/gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (Vertex AI, EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-2.5-flash-image":{"id":"vertex/gemini-2.5-flash-image","name":"Nano Banana (Vertex AI)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3-pro-image":{"id":"vertex/gemini-3-pro-image","name":"Nano Banana Pro (Vertex AI)","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"vertex/gemini-3.1-pro-preview":{"id":"vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview (Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.8-flash@eu":{"id":"vertex/gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (Vertex AI, EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash":{"id":"vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.1-flash-lite":{"id":"vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.1-flash-lite@us":{"id":"vertex/gemini-3.1-flash-lite@us","name":"Gemini 3.1 Flash Lite (Vertex AI, US)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.6-flash@eu":{"id":"vertex/gemini-3.6-flash@eu","name":"Gemini 3.6 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.8-flash@us":{"id":"vertex/gemini-3.8-flash@us","name":"Gemini 3.8 Flash (Vertex AI, US)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash@us":{"id":"vertex/gemini-3.6-flash@us","name":"Gemini 3.6 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.7-flash@us":{"id":"vertex/gemini-3.7-flash@us","name":"Gemini 3.7 Flash (Vertex AI, US)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash":{"id":"vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3.1-flash-image":{"id":"vertex/gemini-3.1-flash-image","name":"Nano Banana 2 (Vertex AI)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"vertex/gemini-3.5-flash-lite":{"id":"vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash-lite@us":{"id":"vertex/gemini-3.5-flash-lite@us","name":"Gemini 3.5 Flash Lite (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash-lite@eu":{"id":"vertex/gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash@us":{"id":"vertex/gemini-3.5-flash@us","name":"Gemini 3.5 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3-flash-preview":{"id":"vertex/gemini-3-flash-preview","name":"Gemini 3 Flash Preview (Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3.8-flash":{"id":"vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.1-flash-lite@eu":{"id":"vertex/gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (Vertex AI, EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.7-flash":{"id":"vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-flash-latest":{"id":"vertex/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash, Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash@eu":{"id":"vertex/gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75,"cache_read":0.35}},"ionos/meta-llama/Llama-3.3-70B-Instruct":{"id":"ionos/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (IONOS)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.74685,"output":0.74685}},"ionos/openai/gpt-oss-120b":{"id":"ionos/openai/gpt-oss-120b","name":"GPT OSS 120B (IONOS)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.17235,"output":0.74685}},"perplexityai/sonar":{"id":"perplexityai/sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":4096},"cost":{"input":1,"output":1}},"perplexityai/sonar-reasoning-pro":{"id":"perplexityai/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"perplexityai/sonar-pro":{"id":"perplexityai/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"perplexityai/sonar-deep-research":{"id":"perplexityai/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"lmstudio":{"id":"lmstudio","env":["LMSTUDIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1234/v1","name":"LMStudio","doc":"https://lmstudio.ai/models","models":{"qwen/qwen3-coder-30b":{"id":"qwen/qwen3-coder-30b","name":"Qwen3 Coder 30B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen/qwen3-30b-a3b-2507":{"id":"qwen/qwen3-30b-a3b-2507","name":"Qwen3 30B A3B 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"lynkr":{"id":"lynkr","env":["LYNKR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:8081/v1","name":"Lynkr","doc":"https://github.com/Fast-Editor/Lynkr","models":{"lynkr-auto":{"id":"lynkr-auto","name":"Lynkr Auto (complexity routing)","description":"Virtual model: Lynkr scores each request on complexity and routes it to the tier model the user configured (local Ollama/llama.cpp for simple requests, configured cloud providers for complex ones).","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}}}}}
\ No newline at end of file
+{"deepinfra":{"id":"deepinfra","env":["DEEPINFRA_API_KEY"],"npm":"@ai-sdk/deepinfra","name":"Deep Infra","doc":"https://deepinfra.com/models","models":{"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.13,"output":0.53,"cache_read":0.033}},"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"meta-llama/Llama-4-Scout-17B-16E-Instruct":{"id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Llama 4 Scout 17B","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.2,"output":0.8}},"XiaomiMiMo/MiMo-V2.6-Pro":{"id":"XiaomiMiMo/MiMo-V2.6-Pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2}},"XiaomiMiMo/MiMo-V2.6-Flash":{"id":"XiaomiMiMo/MiMo-V2.6-Flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.02,"output":0.1}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":2.5,"cache_read":0.05}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":1.1}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"tiers":[{"input":5,"output":15,"cache_read":1,"tier":{"type":"context","size":32000}},{"input":6.25,"output":18.5,"cache_read":1.25,"tier":{"type":"context","size":128000}}]}},"Qwen/Qwen3.8-Max":{"id":"Qwen/Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":1.65,"output":4.951,"cache_read":0.206}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.6}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.1,"output":0.95}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"Qwen/Qwen3-Max":{"id":"Qwen/Qwen3-Max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32000}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128000}}]}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.55}},"Qwen/Qwen3.8-Flash":{"id":"Qwen/Qwen3.8-Flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.113,"output":0.382,"cache_read":0.0141}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.4}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen 3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.45,"output":3,"cache_read":0.22}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.2}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.09,"output":0.18,"cache_read":0.018}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.28,"output":1.1,"cache_read":0.056}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.15,"output":1.15,"cache_read":0.03}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.25,"output":1,"cache_read":0.05}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.75,"output":3.5,"cache_read":0.15}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.85,"output":14.25,"cache_read":0.285}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.68,"output":3.4,"cache_read":0.136}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.75,"output":2.4,"cache_read":0.14}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.6,"output":2.08,"cache_read":0.12}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.5,"output":2,"cache_read":0.1}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.9,"output":4,"cache_read":0.2}},"nvidia/Nemotron-3-Nano-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.4,"output":0.4}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":0.8}},"ByteDance/Seed-2.0-code":{"id":"ByteDance/Seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-pro":{"id":"ByteDance/Seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-mini":{"id":"ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02,"tiers":[{"input":0.2,"output":0.8,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.03,"output":0.14}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.037,"output":0.17}}}},"perplexity-agent":{"id":"perplexity-agent","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.perplexity.ai/v1","name":"Perplexity Agent","doc":"https://docs.perplexity.ai/docs/agent-api/models","models":{"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.25,"output":2.5,"cache_read":0.0625}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"xai/grok-4-1-fast-non-reasoning":{"id":"xai/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32000},"cost":{"input":0.25,"output":2.5}},"moonshot-ai/kimi-k3":{"id":"moonshot-ai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshot-ai/kimi-k2.7-code":{"id":"moonshot-ai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"bailing":{"id":"bailing","env":["BAILING_API_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tbox.cn/api/llm/v1/chat/completions","name":"Bailing","doc":"https://alipaytbox.yuque.com/sxs0ba/ling/intro","models":{"Ring-1T":{"id":"Ring-1T","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}},"Ling-1T":{"id":"Ling-1T","name":"Ling-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}}}},"poe":{"id":"poe","env":["POE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.poe.com/v1","name":"Poe","doc":"https://creator.poe.com/docs/external-applications/openai-compatible-api","models":{"cerebras/qwen3-32b-cs":{"id":"cerebras/qwen3-32b-cs","name":"qwen3-32b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-15","last_updated":"2025-05-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/llama-3.1-8b-cs":{"id":"cerebras/llama-3.1-8b-cs","name":"Llama-3.1-8B-CS","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.1,"output":0.1}},"cerebras/llama-3.3-70b-cs":{"id":"cerebras/llama-3.3-70b-cs","name":"llama-3.3-70b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/gpt-oss-120b-cs":{"id":"cerebras/gpt-oss-120b-cs","name":"GPT-OSS-120B-CS","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.35,"output":0.75}},"cerebras/qwen3-235b-2507-cs":{"id":"cerebras/qwen3-235b-2507-cs","name":"qwen3-235b-2507-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"empiriolabs/deepseek-v4-pro-el":{"id":"empiriolabs/deepseek-v4-pro-el","name":"DeepSeek-V4-Pro-EL","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":1.67,"output":3.33}},"empiriolabs/deepseek-v4-flash-el":{"id":"empiriolabs/deepseek-v4-flash-el","name":"DeepSeek-V4-Flash-EL","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.14,"output":0.28}},"anthropic/claude-haiku-3.5":{"id":"anthropic/claude-haiku-3.5","name":"Claude-Haiku-3.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.68,"output":3.4,"cache_read":0.068,"cache_write":0.85}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude-Opus-4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":32000},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude-Opus-4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-11-21","last_updated":"2025-11-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":64000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude-Opus-4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-sonnet-3.5-june":{"id":"anthropic/claude-sonnet-3.5-june","name":"Claude-Sonnet-3.5-June","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude-Opus-4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.4}},"anthropic/claude-haiku-3":{"id":"anthropic/claude-haiku-3","name":"Claude-Haiku-3","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-09","last_updated":"2024-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.21,"output":1.1,"cache_read":0.021,"cache_write":0.26}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude-Opus-4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.2929,"output":21.4646}},"anthropic/claude-sonnet-3.7":{"id":"anthropic/claude-sonnet-3.7","name":"Claude-Sonnet-3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude-Haiku-4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":64000},"cost":{"input":0.85,"output":4.3,"cache_read":0.085,"cache_write":1.1}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude-Sonnet-4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":32768},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude-Sonnet-4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":64000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-sonnet-3.5":{"id":"anthropic/claude-sonnet-3.5","name":"Claude-Sonnet-3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-06-05","last_updated":"2024-06-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude-Sonnet-4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude-Opus-4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192512,"output":28672},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"elevenlabs/elevenlabs-v2.5-turbo":{"id":"elevenlabs/elevenlabs-v2.5-turbo","name":"ElevenLabs-v2.5-Turbo","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-28","last_updated":"2024-10-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-v3":{"id":"elevenlabs/elevenlabs-v3","name":"ElevenLabs-v3","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-music":{"id":"elevenlabs/elevenlabs-music","name":"ElevenLabs-Music","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-29","last_updated":"2025-08-29","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":2000,"output":0}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"glm-4.6v","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":32768}},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-05-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.96,"output":4.04,"cache_read":0.16}},"novita/kimi-k2-thinking":{"id":"novita/kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":0}},"novita/kimi-k2.5":{"id":"novita/kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"novita/glm-4.7-flash":{"id":"novita/glm-4.7-flash","name":"glm-4.7-flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65500}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"glm-4.7","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"status":"deprecated"},"novita/glm-4.7-n":{"id":"novita/glm-4.7-n","name":"glm-4.7-n","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":0},"cost":{"input":0.27,"output":0.4,"cache_read":0.13}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"minimax-m2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-26","last_updated":"2025-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"lumalabs/ray2":{"id":"lumalabs/ray2","name":"Ray2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ray","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":5000,"output":0}},"poetools/claude-code":{"id":"poetools/claude-code","name":"claude-code","description":"Claude model for careful reasoning, writing, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-27","last_updated":"2025-11-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/imagen-3-fast":{"id":"google/imagen-3-fast","name":"Imagen-3-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-17","last_updated":"2024-10-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4-ultra":{"id":"google/imagen-4-ultra","name":"Imagen-4-Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-24","last_updated":"2025-05-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-deep-research":{"id":"google/gemini-deep-research","name":"gemini-deep-research","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":0},"status":"deprecated","cost":{"input":1.6,"output":9.6}},"google/imagen-4":{"id":"google/imagen-4","name":"Imagen-4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4-fast":{"id":"google/imagen-4-fast","name":"Imagen-4-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/lyria":{"id":"google/lyria","name":"Lyria","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-04","last_updated":"2025-06-04","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/nano-banana":{"id":"google/nano-banana","name":"Nano-Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/veo-3-fast":{"id":"google/veo-3-fast","name":"Veo-3-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.0-flash":{"id":"google/gemini-2.0-flash","name":"Gemini-2.0-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.1,"output":0.42}},"google/veo-3.1":{"id":"google/veo-3.1","name":"Veo-3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/nano-banana-pro":{"id":"google/nano-banana-pro","name":"Nano-Banana-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-2.0-flash-lite":{"id":"google/gemini-2.0-flash-lite","name":"Gemini-2.0-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.052,"output":0.21}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini-3-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini-3.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5152,"output":9.0909,"cache_read":0.1515}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini-2.5-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":32768}],"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.87,"output":7,"cache_read":0.087}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini-2.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-04-26","last_updated":"2025-04-26","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/imagen-3":{"id":"google/imagen-3","name":"Imagen-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3-pro":{"id":"google/gemini-3-pro","name":"Gemini-3-Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":1.6,"output":9.6,"cache_read":0.16}},"google/veo-2":{"id":"google/veo-2","name":"Veo-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo-3.1-fast":{"id":"google/veo-3.1-fast","name":"Veo-3.1-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemma-4-31b":{"id":"google/gemma-4-31b","name":"Gemma-4-31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"google/veo-3":{"id":"google/veo-3","name":"Veo-3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini-2.5-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-06-19","last_updated":"2025-06-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":64000},"cost":{"input":0.07,"output":0.28}},"google/gemini-3.1-pro":{"id":"google/gemini-3.1-pro","name":"Gemini-3.1-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini-3.1-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"xai/grok-3":{"id":"xai/grok-3","name":"Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4.20-multi-agent":{"id":"xai/grok-4.20-multi-agent","name":"Grok-4.20-Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok-4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-fast-reasoning":{"id":"xai/grok-4-fast-reasoning","name":"Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok-4.1-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"xai/grok-4-fast-non-reasoning":{"id":"xai/grok-4-fast-non-reasoning","name":"Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-3-mini":{"id":"xai/grok-3-mini","name":"Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"xai/grok-code-fast-1":{"id":"xai/grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-22","last_updated":"2025-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok-4.1-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"ideogramai/ideogram-v2":{"id":"ideogramai/ideogram-v2","name":"Ideogram-v2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-21","last_updated":"2024-08-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2a-turbo":{"id":"ideogramai/ideogram-v2a-turbo","name":"Ideogram-v2a-Turbo","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2a":{"id":"ideogramai/ideogram-v2a","name":"Ideogram-v2a","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram":{"id":"ideogramai/ideogram","name":"Ideogram","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-04-03","last_updated":"2024-04-03","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"fireworks-ai/kimi-k2.5-fw":{"id":"fireworks-ai/kimi-k2.5-fw","name":"Kimi-K2.5-FW","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":245760,"output":16384},"cost":{"input":0,"output":0}},"runwayml/runway":{"id":"runwayml/runway","name":"Runway","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-11","last_updated":"2024-10-11","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}},"runwayml/runway-gen-4-turbo":{"id":"runwayml/runway-gen-4-turbo","name":"Runway-Gen-4-Turbo","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-09","last_updated":"2025-05-09","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}},"trytako/tako":{"id":"trytako/tako","name":"Tako","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"tako","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":0}},"stabilityai/stablediffusionxl":{"id":"stabilityai/stablediffusionxl","name":"StableDiffusionXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-07-09","last_updated":"2023-07-09","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":200,"output":0}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":14,"cache_read":0.22}},"openai/dall-e-3":{"id":"openai/dall-e-3","name":"DALL-E-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"dall-e","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":800,"output":0}},"openai/sora-2":{"id":"openai/sora-2","name":"Sora-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":27,"output":160}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5-Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":27.2727,"output":163.6364}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4-Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.18,"output":1.1,"cache_read":0.018}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"openai/gpt-4o-search":{"id":"openai/gpt-4o-search","name":"GPT-4o-Search","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9}},"openai/sora-2-pro":{"id":"openai/sora-2-pro","name":"Sora-2-Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5.0505,"output":32.3232,"cache_read":1.2626}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":19,"output":150}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4,"cache_read":0.25}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09,"output":0.36,"cache_read":0.022}},"openai/o3-deep-research":{"id":"openai/o3-deep-research","name":"o3-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":9,"output":36,"cache_read":2.2}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36,"cache_read":0.0045}},"openai/gpt-5.2-instant":{"id":"openai/gpt-5.2-instant","name":"GPT-5.2-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":14,"output":54}},"openai/chatgpt-4o-latest":{"id":"openai/chatgpt-4o-latest","name":"ChatGPT-4o-Latest","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"status":"deprecated","cost":{"input":4.5,"output":14}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":14,"output":110}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5-Turbo-Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-20","last_updated":"2023-09-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":3500,"output":1024},"cost":{"input":1.4,"output":1.8}},"openai/gpt-5.3-codex-spark":{"id":"openai/gpt-5.3-codex-spark","name":"GPT-5.3-Codex-Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5-chat":{"id":"openai/gpt-5-chat","name":"GPT-5-Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":124096,"output":4096},"cost":{"input":0.14,"output":0.54,"cache_read":0.068}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":18,"output":72}},"openai/gpt-5.3-instant":{"id":"openai/gpt-5.3-instant","name":"GPT-5.3-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4o-aug":{"id":"openai/gpt-4o-aug","name":"GPT-4o-Aug","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-21","last_updated":"2024-11-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9,"cache_read":1.1}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3-mini-high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/gpt-5.1-instant":{"id":"openai/gpt-5.1-instant","name":"GPT-5.1-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT-Image-1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-12","last_updated":"2026-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.68,"output":4,"cache_read":0.068}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1-Codex-Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":4.5455,"output":27.2727,"cache_read":0.4545}},"openai/o4-mini-deep-research":{"id":"openai/o4-mini-deep-research","name":"o4-mini-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/gpt-4-classic-0314":{"id":"openai/gpt-4-classic-0314","name":"GPT-4-Classic-0314","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-26","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/gpt-4-classic":{"id":"openai/gpt-4-classic","name":"GPT-4-Classic","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-25","last_updated":"2024-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/gpt-4o-mini-search":{"id":"openai/gpt-4o-mini-search","name":"GPT-4o-mini-Search","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.14,"output":0.54}},"openai/gpt-3.5-turbo-raw":{"id":"openai/gpt-3.5-turbo-raw","name":"GPT-3.5-Turbo-Raw","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":4524,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.36,"output":1.4,"cache_read":0.09}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT-Image-1-Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":9,"output":27}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":140,"output":540}},"topazlabs-co/topazlabs":{"id":"topazlabs-co/topazlabs","name":"TopazLabs","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"topazlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":204,"output":0}}}},"cerebras":{"id":"cerebras","env":["CEREBRAS_API_KEY"],"npm":"@ai-sdk/cerebras","name":"Cerebras","doc":"https://inference-docs.cerebras.ai/models/overview","models":{"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.99,"output":1.49}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.35,"output":0.75}}}},"groq":{"id":"groq","env":["GROQ_API_KEY"],"npm":"@ai-sdk/groq","name":"Groq","doc":"https://console.groq.com/docs/models","models":{"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.59,"output":0.79}},"allam-2-7b":{"id":"allam-2-7b","name":"ALLaM-2-7b","description":"ALLaM-2-7b instruction tuned model by SDAIA","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large V3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Llama 3.1 8B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.08}},"groq/compound-mini":{"id":"groq/compound-mini","name":"Compound Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"groq/compound":{"id":"groq/compound","name":"Compound","description":"General-purpose chat model for instruction following, writing, and analysis","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"meta-llama/llama-prompt-guard-2-86m":{"id":"meta-llama/llama-prompt-guard-2-86m","name":"Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.04,"output":0.04}},"meta-llama/llama-prompt-guard-2-22m":{"id":"meta-llama/llama-prompt-guard-2-22m","name":"Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.03,"output":0.03}},"canopylabs/orpheus-v1-english":{"id":"canopylabs/orpheus-v1-english","name":"Canopy Labs Orpheus V1 English","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"},"canopylabs/orpheus-arabic-saudi":{"id":"canopylabs/orpheus-arabic-saudi","name":"Canopy Labs Orpheus Arabic Saudi","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131042,"output":16384},"cost":{"input":0.8,"output":4}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.6,"output":3,"cache_read":0.3}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"Safety GPT OSS 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta","cost":{"input":0.075,"output":0.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-10-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}}}},"blueclaw":{"id":"blueclaw","env":["BLUECLAW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.blueclaw.network/v1","name":"Blue Claw","doc":"https://blueclaw.network","models":{"Qwen3.6-27B":{"id":"Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"status":"beta"},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta"}}},"zai":{"id":"zai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/paas/v4","name":"Z.AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-4.6v-flash":{"id":"glm-4.6v-flash","name":"GLM-4.6V-Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-5.3-flashx":{"id":"glm-5.3-flashx","name":"GLM-5.3-FlashX","description":"High-speed GLM-5.3-Flash serving option for coding and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}}}},"empiriolabs":{"id":"empiriolabs","env":["EMPIRIOLABS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.empiriolabs.ai/v1","name":"EmpirioLabs AI","doc":"https://docs.empiriolabs.ai","models":{"qwen3-8-max-0902":{"id":"qwen3-8-max-0902","name":"Qwen3.8 Max 0902","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"qwen3-6-plus":{"id":"qwen3-6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.5,"tiers":[{"input":2,"output":6,"cache_read":2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":2}}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.412564,"output":2.475384,"cache_read":0.412564}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.075}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"seed-2-0-lite":{"id":"seed-2-0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.31,"output":2.5,"cache_read":0.31,"tiers":[{"input":0.62,"output":5,"cache_read":0.62,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.424,"output":1.272,"cache_read":0.424}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.7,"output":1.4,"cache_read":0.014}},"seed-2-0-mini":{"id":"seed-2-0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.12,"output":0.5,"cache_read":0.12,"tiers":[{"input":0.24,"output":1,"cache_read":0.24,"tier":{"type":"context","size":128000}}]}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":3}},"glm-4-7-flash":{"id":"glm-4-7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"mimo-v2-6-pro":{"id":"mimo-v2-6-pro","name":"MiMo V2.6 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.435}},"qwen3-7-max":{"id":"qwen3-7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":2.5}},"step-3-5-flash":{"id":"step-3-5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"deepseek-v3-2":{"id":"deepseek-v3-2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.57,"output":1.71,"cache_read":0.57}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.175,"output":4.35,"cache_read":0.018}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.08,"output":5.52,"cache_read":1.08,"tiers":[{"input":2.16,"output":11.04,"cache_read":2.16,"tier":{"type":"context","size":32000}},{"input":2.7,"output":13.8,"cache_read":2.7,"tier":{"type":"context","size":128000}}]}},"seed-2-0-pro":{"id":"seed-2-0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.63,"output":3.79,"cache_read":0.63,"tiers":[{"input":1.26,"output":7.58,"cache_read":1.26,"tier":{"type":"context","size":128000}}]}},"qwen3-5-4b":{"id":"qwen3-5-4b","name":"Qwen3.5 4B","description":"Qwen3.5 4B is a low-cost multimodal reasoning model with 256K context, image and video input, function tools, and structured output.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-02","last_updated":"2026-03-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.04,"output":0.07,"cache_read":0.02}},"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":131072},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"qwen3-5-27b":{"id":"qwen3-5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.086,"output":0.688,"cache_read":0.086,"tiers":[{"input":0.258,"output":2.064,"cache_read":0.258,"tier":{"type":"context","size":128000}}]}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.63,"output":3.13,"cache_read":0.63}},"fugu-ultra-v1-0":{"id":"fugu-ultra-v1-0","name":"Fugu Ultra v1.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":7.5,"output":45,"cache_read":1.5,"tiers":[{"input":15,"output":67.5,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":15,"output":67.5,"cache_read":3}}},"qwen3-5-122b-a10b":{"id":"qwen3-5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.115,"output":0.917,"cache_read":0.115,"tiers":[{"input":0.287,"output":2.294,"cache_read":0.287,"tier":{"type":"context","size":128000}}]}},"qwen3-5-flash":{"id":"qwen3-5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.09,"output":0.368,"cache_read":0.09}},"mimo-v2-6-pro-ultraspeed":{"id":"mimo-v2-6-pro-ultraspeed","name":"MiMo V2.6 Pro UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":4.35}},"glm-5-2":{"id":"glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"mimo-v2-6-flash":{"id":"mimo-v2-6-flash","name":"MiMo V2.6 Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"muse-spark-1-2":{"id":"muse-spark-1-2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"qwen3-7-flash":{"id":"qwen3-7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"tier":{"type":"context","size":256000}}]}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.03}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.07,"output":0.42,"cache_read":0.035}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13,"cache_read":0.045}},"step-3-5-flash-2603":{"id":"step-3-5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.057,"output":0.459,"cache_read":0.057,"tiers":[{"input":0.229,"output":1.835,"cache_read":0.229,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16000},"cost":{"input":0.8939,"output":3.7131,"cache_read":0.1788}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":524288},"cost":{"input":0.225,"output":0.9,"cache_read":0.045,"tiers":[{"input":0.45,"output":1.8,"cache_read":0.09,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.45,"output":1.8,"cache_read":0.09}}},"glm-5-3":{"id":"glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"fugu-ultra-v1-1":{"id":"fugu-ultra-v1-1","name":"Fugu Ultra v1.1","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"kimi-k2-7-code-highspeed":{"id":"kimi-k2-7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.9,"output":8,"cache_read":1.9}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":256000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.4,"tiers":[{"input":1.2,"output":4.8,"cache_read":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":1.2}}},"glm-5-1":{"id":"glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.165,"tiers":[{"input":1.1,"output":3.851,"cache_read":0.22,"tier":{"type":"context","size":32000}}]}},"glm-4-6v-flash":{"id":"glm-4-6v-flash","name":"GLM 4.6V Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0}},"fugu-ultra-v2-0":{"id":"fugu-ultra-v2-0","name":"Fugu Ultra v2.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"mistral-small-4":{"id":"mistral-small-4","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"qwen3-8-omni-flash":{"id":"qwen3-8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":0.94,"cache_read":0.3}},"qwen3-5-plus":{"id":"qwen3-5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.36,"output":2.21,"cache_read":0.36,"tiers":[{"input":1.08,"output":6.62,"cache_read":1.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.08,"output":6.62,"cache_read":1.08}}},"muse-spark-1-3":{"id":"muse-spark-1-3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3,"cache_read":1.65}},"qwen3-8-flash":{"id":"qwen3-8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.16}},"gemma-3-27b":{"id":"gemma-3-27b","name":"Gemma 3 27B","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"qwen3-6-flash":{"id":"qwen3-6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.25,"tiers":[{"input":1,"output":4,"cache_read":1,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1,"output":4,"cache_read":1}}},"seed-2-0-code":{"id":"seed-2-0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.4,"tiers":[{"input":0.8,"output":4.8,"cache_read":0.8,"tier":{"type":"context","size":128000}}]}},"qwen3-8-27b":{"id":"qwen3-8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.17,"output":0.5,"cache_read":0.08}},"muse-spark-1-1":{"id":"muse-spark-1-1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"qwen3-6-max-preview":{"id":"qwen3-6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88,"cache_read":1.31,"tiers":[{"input":1.97,"output":11.82,"cache_read":1.97,"tier":{"type":"context","size":128000}}]}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":1.032,"cache_read":0.172,"tiers":[{"input":0.43,"output":2.58,"cache_read":0.43,"tier":{"type":"context","size":128000}}]}},"glm-4-5-flash":{"id":"glm-4-5-flash","name":"GLM 4.5 Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":98304},"cost":{"input":0,"output":0}},"gemma-4-26b-a4b":{"id":"gemma-4-26b-a4b","name":"Gemma 4 26B-A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.29,"cache_read":0.025}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}}}},"sensenova":{"id":"sensenova","env":["SENSENOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token.sensenova.cn/v1","name":"SenseNova (China)","doc":"https://platform.sensenova.cn/docs","models":{"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"sensenova-6.8-flash-lite":{"id":"sensenova-6.8-flash-lite","name":"SenseNova 6.8 Flash Lite","description":"SenseNova lightweight multimodal agent model for real-world complex tasks, data analysis, and complex information presentation","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}}}},"alibaba-token-plan":{"id":"alibaba-token-plan","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/token-plan-overview","models":{"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"cloudflare-workers-ai":{"id":"cloudflare-workers-ai","env":["CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1","name":"Cloudflare Workers AI","doc":"https://developers.cloudflare.com/workers-ai/models/","models":{"@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"id":"@cf/meta/llama-3.3-70b-instruct-fp8-fast","name":"Llama 3.3 70B Instruct fp8 Fast","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.293,"output":2.253}},"@cf/meta/llama-guard-3-8b":{"id":"@cf/meta/llama-guard-3-8b","name":"Llama Guard 3 8B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.484,"output":0.03}},"@cf/meta/llama-3.1-8b-instruct-fp8":{"id":"@cf/meta/llama-3.1-8b-instruct-fp8","name":"Llama 3.1 8B Instruct fp8","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.152,"output":0.287}},"@cf/meta/llama-3.2-11b-vision-instruct":{"id":"@cf/meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.0485,"output":0.676}},"@cf/meta/llama-3.2-1b-instruct":{"id":"@cf/meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":60000},"cost":{"input":0.027,"output":0.201}},"@cf/meta/llama-4-scout-17b-16e-instruct":{"id":"@cf/meta/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":16384},"cost":{"input":0.27,"output":0.85}},"@cf/meta/llama-3.2-3b-instruct":{"id":"@cf/meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.0509,"output":0.335}},"@cf/google/gemma-4-26b-a4b-it":{"id":"@cf/google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1,"output":0.3}},"@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma Sea Lion V4 27B It","description":"Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/ibm-granite/granite-4.0-h-micro":{"id":"@cf/ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 H Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.017,"output":0.112}},"@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"id":"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b","name":"Deepseek R1 Distill Qwen 32B","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.497,"output":4.881}},"@cf/mistralai/mistral-small-3.1-24b-instruct":{"id":"@cf/mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/moonshotai/kimi-k2.6":{"id":"@cf/moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"@cf/moonshotai/kimi-k2.7-code":{"id":"@cf/moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"@cf/zai-org/glm-5.3-flash":{"id":"@cf/zai-org/glm-5.3-flash","name":"Glm 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"@cf/zai-org/glm-4.7-flash":{"id":"@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"@cf/zai-org/glm-5.2":{"id":"@cf/zai-org/glm-5.2","name":"Glm 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/zai-org/glm-5.3":{"id":"@cf/zai-org/glm-5.3","name":"Glm 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/nvidia/nemotron-3-120b-a12b":{"id":"@cf/nvidia/nemotron-3-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5}},"@cf/qwen/qwen3.8-27b":{"id":"@cf/qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":3.2,"cache_read":0.05}},"@cf/qwen/qwq-32b":{"id":"@cf/qwen/qwq-32b","name":"Qwq 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.66,"output":1}},"@cf/qwen/qwen3-30b-a3b-fp8":{"id":"@cf/qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3b fp8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.0509,"output":0.335}},"@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.66,"output":1}},"@cf/openai/gpt-oss-20b":{"id":"@cf/openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"@cf/openai/gpt-oss-120b":{"id":"@cf/openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.35,"output":0.75}}}},"poolside":{"id":"poolside","env":["POOLSIDE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.poolside.ai/v1","name":"Poolside","doc":"https://platform.poolside.ai","models":{"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-m.1":{"id":"poolside/laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"nano-gpt":{"id":"nano-gpt","env":["NANO_GPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://nano-gpt.com/api/v1","name":"NanoGPT","doc":"https://docs.nano-gpt.com","models":{"deepseek-reasoner-cheaper":{"id":"deepseek-reasoner-cheaper","name":"Deepseek R1 Cheaper","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"nano-gpt-help":{"id":"nano-gpt-help","name":"NanoGPT Help","description":"Text-only NanoGPT support assistant. Questions are processed by the Help inference provider; do not paste secrets or account credentials. Covers the website, models, API, pricing, memory, media generation, and support.","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6000,"input":6000,"output":512},"cost":{"input":0,"output":0}},"doubao-seed-2-0-lite-260215":{"id":"doubao-seed-2-0-lite-260215","name":"Doubao Seed 2.0 Lite","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.1462,"output":0.8738,"cache_read":0.0731}},"doubao-seed-2-0-mini-260215":{"id":"doubao-seed-2-0-mini-260215","name":"Doubao Seed 2.0 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.0493,"output":0.4845,"cache_read":0.02465}},"fastgpt":{"id":"fastgpt","name":"Web Answer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":7.5,"output":7.5}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Doubao Seed 2.0 Pro","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.876,"cache_read":0.391}},"ernie-5.0-thinking-preview":{"id":"ernie-5.0-thinking-preview","name":"Ernie 5.0 Thinking Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":3.5,"cache_read":0.5}},"gemini-2.5-flash-lite-preview-09-2025-thinking":{"id":"gemini-2.5-flash-lite-preview-09-2025-thinking","name":"Gemini 2.5 Flash Lite Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"gemma-4-12b-it-station-keeper":{"id":"gemma-4-12b-it-station-keeper","name":"Gemma 4 12B StationKeeper","description":"Gemma 4 12B StationKeeper is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"glm-4-air-0111":{"id":"glm-4-air-0111","name":"GLM 4 Air 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-11","last_updated":"2025-01-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":0.1394,"output":0.1394,"cache_read":0.0697}},"glm-4.1v-thinking-flashx":{"id":"glm-4.1v-thinking-flashx","name":"GLM 4.1V Thinking FlashX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek Chat 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.77,"cache_read":0.135}},"venice-uncensored":{"id":"venice-uncensored","name":"Venice Uncensored","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"venice","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-01","last_updated":"2025-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.4}},"Gemma-4-31B-MeroMero-v2:thinking":{"id":"Gemma-4-31B-MeroMero-v2:thinking","name":"Gemma 4 31B MeroMero v2 Thinking","description":"Gemma 4 31B MeroMero v2 with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemini-2.0-pro-reasoner":{"id":"gemini-2.0-pro-reasoner","name":"Gemini 2.0 Pro Reasoner","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-05","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1.292,"output":4.998,"cache_read":0.323}},"Meta-Llama-3-1-8B-Instruct-FP8":{"id":"Meta-Llama-3-1-8B-Instruct-FP8","name":"Llama 3.1 8B (decentralized)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.02,"output":0.03,"cache_read":0.01}},"glm-4-plus-0111":{"id":"glm-4-plus-0111","name":"GLM 4 Plus 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":9.996,"output":9.996,"cache_read":4.998}},"holo3-35b-a3b:thinking":{"id":"holo3-35b-a3b:thinking","name":"Holo3-35B-A3B Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"gemma-4-26b-a4b-it-moonlight":{"id":"gemma-4-26b-a4b-it-moonlight","name":"Moonlight Dusk","description":"Moonlight Dusk is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"phi-4-mini-instruct":{"id":"phi-4-mini-instruct","name":"Phi 4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"auto-model-premium":{"id":"auto-model-premium","name":"Auto model (Premium)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"gemma-4-31b-it-darkidol":{"id":"gemma-4-31b-it-darkidol","name":"DarkIdol","description":"DarkIdol is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"Gemini 2.5 Flash Lite Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"Gemini 2.5 Flash Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemma-4-26b-a4b-it-opusdistill":{"id":"gemma-4-26b-a4b-it-opusdistill","name":"Opus Distill","description":"Opus Distill is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"auto-model-standard":{"id":"auto-model-standard","name":"Auto model (Standard)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat 2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"gemini-2.5-pro-preview-03-25":{"id":"gemini-2.5-pro-preview-03-25","name":"Gemini 2.5 Pro Preview 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"agnes-3.0-flash":{"id":"agnes-3.0-flash","name":"Agnes 3.0 Flash","description":"Agnes 3.0 Flash is a low-cost model for coding, tool use, and multi-turn agent tasks. It supports text and image input, optional thinking, and a 512K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.005}},"Gemma-4-31B-Queen":{"id":"Gemma-4-31B-Queen","name":"Gemma 4 31B Queen","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"auto-model":{"id":"auto-model","name":"Auto model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":0,"output":0}},"qvq-max":{"id":"qvq-max","name":"Qwen: QvQ Max","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-28","last_updated":"2025-03-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":1.2,"output":4.8,"cache_read":0.6}},"asi1-mini":{"id":"asi1-mini","name":"ASI1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":1,"cache_read":0.5}},"gemini-2.5-pro-exp-03-25":{"id":"gemini-2.5-pro-exp-03-25","name":"Gemini 2.5 Pro Experimental 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"gemma-4-31b-it-isometry":{"id":"gemma-4-31b-it-isometry","name":"Isometry","description":"Isometry is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"celeris-1":{"id":"celeris-1","name":"Celeris 1","description":"Celeris 1 is a diffusion language model built for ultra-low-latency classification, extraction, judging, query rewriting, and other short structured responses.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-07-25","last_updated":"2026-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":2,"output":6,"cache_read":1}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"gemma-4-26b-a4b-it-darksoul":{"id":"gemma-4-26b-a4b-it-darksoul","name":"Dark Soul","description":"Dark Soul is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"glm-4.1v-thinking-flash":{"id":"glm-4.1v-thinking-flash","name":"GLM 4.1V Thinking Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"Gemma-4-31B-GarnetV2":{"id":"Gemma-4-31B-GarnetV2","name":"Gemma 4 31B Garnet V2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemini-2.5-flash-nothinking":{"id":"gemini-2.5-flash-nothinking","name":"Gemini 2.5 Flash (No Thinking)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"Gemma-4-31B-MeroMero-v2":{"id":"Gemma-4-31B-MeroMero-v2","name":"Gemma 4 31B MeroMero v2","description":"Gemma 4 31B MeroMero v2 is a LoRA finetune for emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"holo3-35b-a3b":{"id":"holo3-35b-a3b","name":"Holo3-35B-A3B","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"claw-medium":{"id":"claw-medium","name":"Claw Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"gemma-4-26b-a4b-it-shadowsiren":{"id":"gemma-4-26b-a4b-it-shadowsiren","name":"Shadow Siren","description":"Shadow Siren is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemini-2.0-pro-exp-02-05":{"id":"gemini-2.0-pro-exp-02-05","name":"Gemini 2.0 Pro 0205","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.989,"output":7.956,"cache_read":0.49725}},"claw-high":{"id":"claw-high","name":"Claw High","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-chat":{"id":"deepseek-chat","name":"DeepSeek V3/Deepseek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"Gemma-4-26B-A4B-MeroMero:thinking":{"id":"Gemma-4-26B-A4B-MeroMero:thinking","name":"Gemma 4 26B A4B MeroMero Thinking","description":"Gemma 4 26B A4B MeroMero with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"claw-low":{"id":"claw-low","name":"Claw Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.1343,"output":0.3349,"cache_read":0.06715}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Doubao Seed 2.0 Code Preview","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.893,"cache_read":0.391}},"hermes-low":{"id":"hermes-low","name":"Hermes Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"gemma-4-31b-it-gemsicle":{"id":"gemma-4-31b-it-gemsicle","name":"Gemsicle","description":"Gemsicle is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"mercury-coder-small":{"id":"mercury-coder-small","name":"Mercury Coder Small","description":"Model by Inception AI. A diffusion large language model that runs incredibly quickly (500+ tokens/second) while matching Claude 3.5 Haiku and GPT-4o-mini. 1st in speed on Copilot arena, and matching 2nd in quality.","family":"mercury","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":0.375}},"ernie-x1.1-preview":{"id":"ernie-x1.1-preview","name":"ERNIE X1.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"hermes-high":{"id":"hermes-high","name":"Hermes High","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"Gemini 2.5 Flash 0520","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"glm-4-long":{"id":"glm-4-long","name":"GLM-4 Long","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":4096},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"pokee-isaac":{"id":"pokee-isaac","name":"Pokee-Isaac 28B","description":"Pokee-Isaac is a 28B agentic model with a roughly 10-million-token context window, function calling, and OpenAI-compatible structured output. Pokee bills in $0.01 increments, rounding each non-zero request up to the next cent.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":60000},"cost":{"input":0.15,"output":1,"cache_read":0.075}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.1394,"output":1.3328,"cache_read":0.0697}},"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled":{"id":"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled","name":"Gemma 4 31B Claude 4.6 Opus Reasoning Distilled","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"claude","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.0306}},"Qwen3.5-27B-BlueStar-v3-Derestricted":{"id":"Qwen3.5-27B-BlueStar-v3-Derestricted","name":"Qwen3.5 27B BlueStar v3 Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemma-4-31b-it-novelist":{"id":"gemma-4-31b-it-novelist","name":"Novelist","description":"Novelist is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"ernie-5.1:thinking":{"id":"ernie-5.1:thinking","name":"ERNIE 5.1 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"gemma-4-12b-it-semancer":{"id":"gemma-4-12b-it-semancer","name":"Gemma 4 12B Semancer","description":"Gemma 4 12B Semancer is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"mistral-code-latest":{"id":"mistral-code-latest","name":"Mistral Code Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"Qwen3.5-27B-Queen-Derestricted":{"id":"Qwen3.5-27B-Queen-Derestricted","name":"Qwen3.5 27B Queen Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemma-4-31b-it-fabled":{"id":"gemma-4-31b-it-fabled","name":"Fabled","description":"Fabled is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"Gemma-4-26B-A4B-MeroMero":{"id":"Gemma-4-26B-A4B-MeroMero","name":"Gemma 4 26B A4B MeroMero","description":"Gemma 4 26B A4B MeroMero is an NVFP4 multimodal mixture-of-experts fine-tune for emotive dialogue, relationship scenes, creative writing, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"Gemma-4-31B-Cognitive-Unshackled":{"id":"Gemma-4-31B-Cognitive-Unshackled","name":"Gemma 4 31B Cognitive Unshackled","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"universal-summarizer":{"id":"universal-summarizer","name":"Universal Summarizer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":30,"output":30}},"gemma-4-12b-it":{"id":"gemma-4-12b-it","name":"Gemma 4 12B Instruct","description":"Compact Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"gemini-exp-1206":{"id":"gemini-exp-1206","name":"Gemini 2.0 Pro 1206","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.258,"output":4.998,"cache_read":0.629}},"gemini-2.5-flash-preview-09-2025-thinking":{"id":"gemini-2.5-flash-preview-09-2025-thinking","name":"Gemini 2.5 Flash Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"doubao-1.5-pro-256k":{"id":"doubao-1.5-pro-256k","name":"Doubao 1.5 Pro 256k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.799,"output":1.445,"cache_read":0.3995}},"gemma-4-31b-it-garnet":{"id":"gemma-4-31b-it-garnet","name":"Garnet","description":"Garnet is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemini-2.5-pro-preview-05-06":{"id":"gemini-2.5-pro-preview-05-06","name":"Gemini 2.5 Pro Preview 0506","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-06","last_updated":"2025-05-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Cohere Command A (08/2025)","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"longcat-2.0:thinking":{"id":"longcat-2.0:thinking","name":"LongCat 2.0 Thinking","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"deepseek-r1-sambanova":{"id":"deepseek-r1-sambanova","name":"DeepSeek R1 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":4.998,"output":6.987,"cache_read":2.499}},"gemini-2.5-flash-preview-04-17":{"id":"gemini-2.5-flash-preview-04-17","name":"Gemini 2.5 Flash Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"input":64000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"Gemini 2.5 Flash Lite Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"gemini-2.5-flash-preview-04-17:thinking":{"id":"gemini-2.5-flash-preview-04-17:thinking","name":"Gemini 2.5 Flash Preview Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"Gemini 2.5 Pro Preview 0605","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"deepseek-chat-cheaper":{"id":"deepseek-chat-cheaper","name":"DeepSeek V3/Chat Cheaper","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"GLM-4.6-Derestricted-v5":{"id":"GLM-4.6-Derestricted-v5","name":"GLM 4.6 Derestricted v5","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.4,"output":1.5,"cache_read":0.2}},"doubao-1.5-vision-pro-32k":{"id":"doubao-1.5-vision-pro-32k","name":"Doubao 1.5 Vision Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.459,"output":1.377,"cache_read":0.2295}},"glm-z1-airx":{"id":"glm-z1-airx","name":"GLM Z1 AirX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"auto-model-basic":{"id":"auto-model-basic","name":"Auto model (Basic)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"qwen3-vl-235b-a22b-instruct-original":{"id":"qwen3-vl-235b-a22b-instruct-original","name":"Qwen3 VL 235B A22B Instruct Original","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"gemma-4-26b-a4b-it-luminous":{"id":"gemma-4-26b-a4b-it-luminous","name":"Luminous Mirror","description":"Luminous Mirror is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":4096},"cost":{"input":0.054,"output":0.2124,"cache_read":0.0336}},"gemma-4-31b-it-gembrain":{"id":"gemma-4-31b-it-gembrain","name":"Gembrain","description":"Gembrain is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"phi-4-multimodal-instruct":{"id":"phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.07,"output":0.11,"cache_read":0.035}},"gemini-2.5-flash-preview-05-20:thinking":{"id":"gemini-2.5-flash-preview-05-20:thinking","name":"Gemini 2.5 Flash 0520 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"gemma-4-26b-a4b-it-musica":{"id":"gemma-4-26b-a4b-it-musica","name":"Musica","description":"Musica is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemma-4-26b-a4b-it-chimerax":{"id":"gemma-4-26b-a4b-it-chimerax","name":"Chimera X","description":"Chimera X is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"hermes-medium":{"id":"hermes-medium","name":"Hermes Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"doubao-seed-1-6-250615":{"id":"doubao-seed-1-6-250615","name":"Doubao Seed 1.6","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.204,"output":0.51,"cache_read":0.102}},"Gemma-4-31B-DarkIdol":{"id":"Gemma-4-31B-DarkIdol","name":"Gemma 4 31B DarkIdol","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"ernie-5.1":{"id":"ernie-5.1","name":"ERNIE 5.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"doubao-seed-1-6-flash-250615":{"id":"doubao-seed-1-6-flash-250615","name":"Doubao Seed 1.6 Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.0374,"output":0.374,"cache_read":0.0187}},"mistral-code-agent-latest":{"id":"mistral-code-agent-latest","name":"Mistral Code Agent Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"kimi-k2-instruct-fast":{"id":"kimi-k2-instruct-fast","name":"Kimi K2 0711 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-15","last_updated":"2025-07-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"featherless-ai/Qwerky-72B":{"id":"featherless-ai/Qwerky-72B","name":"Qwerky 72B","description":"General-purpose chat model for instruction following, writing, and analysis","family":"qwerky","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"stealth/space-bunny-alpha":{"id":"stealth/space-bunny-alpha","name":"Space Bunny Alpha","description":"Space Bunny Alpha is an anonymous stealth preview model for coding and multimodal tasks. It accepts text, images, and video, supports tool calling and structured output, and always reasons with adjustable effort across a 1M-token context window.","family":"alpha","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":524288},"cost":{"input":0.05,"output":0.15}},"unsloth/gemma-3-27b-it":{"id":"unsloth/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":96000},"cost":{"input":0.2992,"output":0.2992,"cache_read":0.1496}},"unsloth/gemma-3-12b-it":{"id":"unsloth/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.272,"output":0.272,"cache_read":0.136}},"unsloth/gemma-3-4b-it":{"id":"unsloth/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"ByteDance Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.25,"output":2,"cache_read":0.125}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed 2.1 Turbo","description":"ByteDance Seed 2.1 Turbo is a multimodal model for coding and long-horizon agent workflows, including end-to-end software delivery and multi-step task execution. It supports text, image, and video input with a 262k context window.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"ByteDance Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.25}},"poolside/laguna-s-2.1:thinking":{"id":"poolside/laguna-s-2.1:thinking","name":"Laguna S 2.1 Thinking","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"abliteration-ai/abliterated-model-large-v2":{"id":"abliteration-ai/abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"Abliteration.ai's default large text reasoning model is derived from GLM-5.3 for harder reasoning and evaluation workloads, with automatic prompt caching and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":3,"output":5,"cache_read":0.3}},"abliteration-ai/abliterated-model":{"id":"abliteration-ai/abliterated-model","name":"Abliterated Model","description":"Abliteration.ai's multimodal reasoning model supports text and image input, structured output, automatic prompt caching, and a 262K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":262134},"cost":{"input":1,"output":3,"cache_read":0.1}},"abliteration-ai/abliterated-model-large":{"id":"abliteration-ai/abliterated-model-large","name":"Abliterated Model Large","description":"Abliteration.ai's large text reasoning model is derived from GLM-5.2 and supports native tool calling, structured output, automatic prompt caching, and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":3,"output":5,"cache_read":0.3}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"Ternary Bonsai 2 27B","description":"Ternary Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.5,"cache_read":0.0375}},"anthropic/claude-opus-4.5:thinking":{"id":"anthropic/claude-opus-4.5:thinking","name":"Claude 4.5 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4.5:thinking":{"id":"anthropic/claude-sonnet-4.5:thinking","name":"Claude Sonnet 4.5 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6:thinking:low":{"id":"anthropic/claude-opus-4.6:thinking:low","name":"Claude 4.6 Opus Thinking Low","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude 4.6 Opus","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude 4.7 Opus","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4:thinking:64000":{"id":"anthropic/claude-sonnet-4:thinking:64000","name":"Claude 4 Sonnet Thinking (64K)","description":"Claude 4 Sonnet with maximum thinking budget (64,000 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1:thinking:1024":{"id":"anthropic/claude-opus-4.1:thinking:1024","name":"Claude 4.1 Opus Thinking (1K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-haiku-latest":{"id":"anthropic/claude-haiku-latest","name":"Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4:thinking:8192":{"id":"anthropic/claude-sonnet-4:thinking:8192","name":"Claude 4 Sonnet Thinking (8K)","description":"Claude 4 Sonnet with reduced thinking budget (8,192 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-4:thinking:1024":{"id":"anthropic/claude-sonnet-4:thinking:1024","name":"Claude 4 Sonnet Thinking (1K)","description":"Claude 4 Sonnet with minimal thinking budget (1,024 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6:thinking":{"id":"anthropic/claude-opus-4.6:thinking","name":"Claude 4.6 Opus Thinking","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4.6:thinking:medium":{"id":"anthropic/claude-opus-4.6:thinking:medium","name":"Claude 4.6 Opus Thinking Medium","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Compatibility alias for Claude Fable.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4:thinking:32768":{"id":"anthropic/claude-sonnet-4:thinking:32768","name":"Claude 4 Sonnet Thinking (32K)","description":"Claude 4 Sonnet with extended thinking budget (32,768 tokens).","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.1:thinking:8192":{"id":"anthropic/claude-opus-4.1:thinking:8192","name":"Claude 4.1 Opus Thinking (8K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-haiku-4.5:thinking":{"id":"anthropic/claude-haiku-4.5:thinking","name":"Claude Haiku 4.5 Thinking","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4:thinking:1024":{"id":"anthropic/claude-opus-4:thinking:1024","name":"Claude 4 Opus Thinking (1K)","description":"Claude 4 Opus with minimal thinking budget (1,024 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.8:thinking":{"id":"anthropic/claude-opus-4.8:thinking","name":"Claude Opus 4.8 Thinking","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude 4 Sonnet","description":"Claude 4 Sonnet by Anthropic. A new generation model with improved capabilities, especially on programming and development. NOTE: Inputs > 200k tokens are charged at 2x input, 1.5x output rate.","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6:thinking":{"id":"anthropic/claude-sonnet-4.6:thinking","name":"Claude Sonnet 4.6 Thinking","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4:thinking:32768":{"id":"anthropic/claude-opus-4:thinking:32768","name":"Claude 4 Opus Thinking (32K)","description":"Claude 4 Opus with extended thinking budget (32,768 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.6:thinking:max":{"id":"anthropic/claude-opus-4.6:thinking:max","name":"Claude 4.6 Opus Thinking Max","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-5:thinking":{"id":"anthropic/claude-sonnet-5:thinking","name":"Claude Sonnet 5 Thinking","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.7:thinking":{"id":"anthropic/claude-opus-4.7:thinking","name":"Claude 4.7 Opus Thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-opus-4.1:thinking":{"id":"anthropic/claude-opus-4.1:thinking","name":"Claude 4.1 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-sonnet-4:thinking":{"id":"anthropic/claude-sonnet-4:thinking","name":"Claude 4 Sonnet Thinking","description":"Anthropic's Claude 4 Sonnet with the ability to show its thinking process step by step.","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4:thinking":{"id":"anthropic/claude-opus-4:thinking","name":"Claude 4 Opus Thinking","description":"Anthropic's Claude 4 Opus with the ability to show its thinking process step by step.","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4:thinking:8192":{"id":"anthropic/claude-opus-4:thinking:8192","name":"Claude 4 Opus Thinking (8K)","description":"Claude 4 Opus with reduced thinking budget (8,192 tokens).","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude 4 Opus","description":"Claude 4 Opus by Anthropic. The premium version of the new Claude models. A new generation model with improved capabilities, especially on programming and development.","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"anthropic/claude-opus-4.1:thinking:32768":{"id":"anthropic/claude-opus-4.1:thinking:32768","name":"Claude 4.1 Opus Thinking (32K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Cohere: Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":2.856,"output":14.246,"cache_read":1.428}},"deepseek/deepseek-v4-flash:thinking":{"id":"deepseek/deepseek-v4-flash:thinking","name":"DeepSeek V4 Flash (Thinking)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.13,"output":0.52,"cache_read":0.006}},"deepseek/deepseek-v4.1-flash:thinking":{"id":"deepseek/deepseek-v4.1-flash:thinking","name":"DeepSeek V4.1 Flash Thinking","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.13,"output":0.52,"cache_read":0.006}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v3.2:thinking":{"id":"deepseek/deepseek-v3.2:thinking","name":"DeepSeek V3.2 Thinking","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek/deepseek-latest":{"id":"deepseek/deepseek-latest","name":"DeepSeek Latest","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-pro:thinking":{"id":"deepseek/deepseek-v4-pro:thinking","name":"DeepSeek V4 Pro (Thinking)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v4-pro-0813:thinking":{"id":"deepseek/deepseek-v4-pro-0813:thinking","name":"DeepSeek V4 Pro 0813 Thinking","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-flash-0731:thinking":{"id":"deepseek/deepseek-v4-flash-0731:thinking","name":"DeepSeek V4 Flash 0731 (Thinking)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Compatibility alias that routes to the newest dated DeepSeek V4 Flash release. Currently routes to DeepSeek V4 Flash 0731. ⚠️ This route goes directly to DeepSeek, so privacy and logging guarantees are limited.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"baseten/Kimi-K2-Instruct-FP4":{"id":"baseten/Kimi-K2-Instruct-FP4","name":"Kimi K2 0711 Instruct FP4","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"lightonai/LightOnOCR-2-1B":{"id":"lightonai/LightOnOCR-2-1B","name":"LightOnOCR 2","description":"LightOnOCR 2 hosted by IONOS in Berlin, Germany. Zero data retention.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.1785,"output":0.3465}},"abacusai/Dracarys-72B-Instruct":{"id":"abacusai/Dracarys-72B-Instruct","name":"Llama 3.1 70B Dracarys 2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"shisa-ai/shisa-v2.1-llama3.3-70b":{"id":"shisa-ai/shisa-v2.1-llama3.3-70b","name":"Shisa V2.1 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"shisa-ai/shisa-v2-llama3.3-70b":{"id":"shisa-ai/shisa-v2-llama3.3-70b","name":"Shisa V2 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"Steelskull/L3.3-MS-Nevoria-70b":{"id":"Steelskull/L3.3-MS-Nevoria-70b","name":"Steelskull Nevoria 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-Cu-Mai-R1-70b":{"id":"Steelskull/L3.3-Cu-Mai-R1-70b","name":"Llama 3.3 70B Cu Mai","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-Nevoria-R1-70b":{"id":"Steelskull/L3.3-Nevoria-R1-70b","name":"Steelskull Nevoria R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-Electra-R1-70b":{"id":"Steelskull/L3.3-Electra-R1-70b","name":"Steelskull Electra R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.69989,"output":0.69989,"cache_read":0.349945}},"tencent/hy3":{"id":"tencent/hy3","name":"Tencent Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":128000},"cost":{"input":0.066,"output":0.26,"cache_read":0.029}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"anthracite-org/magnum-v2-72b":{"id":"anthracite-org/magnum-v2-72b","name":"Magnum V2 72B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":328000,"input":328000,"output":65536},"cost":{"input":0.085,"output":0.46,"cache_read":0.0425}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.05,"output":0.23,"cache_read":0.025}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8b Instruct","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.0544,"output":0.085,"cache_read":0.0272}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3b Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-09-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.0306,"output":0.0493,"cache_read":0.0153}},"LLM360/K2-Think":{"id":"LLM360/K2-Think","name":"K2-Think","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"deepcogito/cogito-v1-preview-qwen-32B":{"id":"deepcogito/cogito-v1-preview-qwen-32B","name":"Cogito v1 Preview Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-10","last_updated":"2025-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":1.8,"output":1.8,"cache_read":0.9}},"GalrionSoftworks/MN-LooseCannon-12B-v1":{"id":"GalrionSoftworks/MN-LooseCannon-12B-v1","name":"MN-LooseCannon-12B-v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"meganova-ai/manta-mini-1.0":{"id":"meganova-ai/manta-mini-1.0","name":"Manta Mini 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-flash-1.0":{"id":"meganova-ai/manta-flash-1.0","name":"Manta Flash 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":16384},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-pro-1.0":{"id":"meganova-ai/manta-pro-1.0","name":"Manta Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":0.06,"output":0.5,"cache_read":0.03}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.15,"output":1.5,"cache_read":0.075}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"z-ai/glm-4.6-original":{"id":"z-ai/glm-4.6-original","name":"GLM 4.6 Original","description":"GLM-4.6, Zhipu's flagship text model with 256K context window and advanced reasoning capabilities. Direct via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015}},"z-ai/glm-5.3-flash-uncensored":{"id":"z-ai/glm-5.3-flash-uncensored","name":"GLM 5.3 Flash Uncensored","description":"GLM 5.3 Flash Uncensored is an uncensored fine-tune of the efficient 320B mixture-of-experts reasoning model, built for unrestricted chat, creative writing, coding, agentic work, tool use, and long-context tasks.","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.7-flash:thinking":{"id":"z-ai/glm-4.7-flash:thinking","name":"GLM 4.7 Flash Thinking","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/glm-5-original:thinking":{"id":"z-ai/glm-5-original:thinking","name":"GLM 5 Original Thinking","description":"GLM-5 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3:thinking":{"id":"z-ai/glm-5.3:thinking","name":"GLM 5.3 Thinking","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/GLM-4.5-Air":{"id":"z-ai/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/GLM-4.6-turbo:thinking":{"id":"z-ai/GLM-4.6-turbo:thinking","name":"GLM 4.6 Turbo (Thinking)","description":"GLM 4.6 Turbo with thinking mode enabled for enhanced reasoning; shows internal reasoning and supports long context.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/glm-4.7:thinking":{"id":"z-ai/glm-4.7:thinking","name":"GLM 4.7 Thinking","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-5.1:thinking":{"id":"z-ai/glm-5.1:thinking","name":"GLM 5.1 Thinking","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-4.6v-original":{"id":"z-ai/glm-4.6v-original","name":"GLM 4.6V Original","description":"GLM-4.6V scales its context window to 128k tokens in training, and achieves SoTA performance in visual understanding among models of similar parameter scales. Integrates native Function Calling capabilities, bridging 'visual perception' and 'executable action' for multimodal agents. Direct via Z-AI (Zhipu).","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.6,"output":0.9,"cache_read":0.3}},"z-ai/glm-5v-turbo:thinking":{"id":"z-ai/glm-5v-turbo:thinking","name":"GLM 5V Turbo Thinking","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/GLM-4.6-turbo":{"id":"z-ai/GLM-4.6-turbo","name":"GLM 4.6 Turbo","description":"Fast variant of GLM 4.6 for general chat, coding, and analysis with improved latency and strong reasoning.","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/GLM-4.5-Air:thinking":{"id":"z-ai/GLM-4.5-Air:thinking","name":"GLM 4.5 Air (Thinking)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-5.2:thinking":{"id":"z-ai/glm-5.2:thinking","name":"GLM 5.2 Thinking","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-5:thinking":{"id":"z-ai/glm-5:thinking","name":"GLM 5 Thinking","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/GLM-4.5:thinking":{"id":"z-ai/GLM-4.5:thinking","name":"GLM 4.5 (Thinking)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-4.6:thinking":{"id":"z-ai/glm-4.6:thinking","name":"GLM 4.6 Thinking","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.5v:thinking":{"id":"z-ai/glm-4.5v:thinking","name":"GLM 4.5V Thinking","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/glm-4.7-original":{"id":"z-ai/glm-4.7-original","name":"GLM 4.7 Original","description":"GLM-4.7 is a next-gen GLM series text model with stronger reasoning, long-context chat, and reliable tool use. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5-original":{"id":"z-ai/glm-5-original","name":"GLM 5 Original","description":"GLM-5 is Zhipu's latest flagship model with advanced reasoning and instruction following. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-original:thinking":{"id":"z-ai/glm-4.7-original:thinking","name":"GLM 4.7 Original Thinking","description":"GLM-4.7 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3-flash-cybersecurity":{"id":"z-ai/glm-5.3-flash-cybersecurity","name":"GLM 5.3 Flash Cybersecurity","description":"GLM 5.3 Flash Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports always-on reasoning, image understanding, tool calling, and a 1,048,576-token context window.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":32768},"cost":{"input":0.15,"output":0.5,"cache_read":0.075}},"z-ai/glm-latest":{"id":"z-ai/glm-latest","name":"GLM Latest","description":"Compatibility alias that routes to the newest thinking GLM model. Currently routes to GLM 5.2 Thinking.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Inference.net's 3B-parameter HTML-to-JSON extraction model, focused on accuracy for complex schemas and long web pages. It turns HTML into typed, structured data for web scraping and product catalog ingestion, with a 128K-token context window. Supply HTML in the user message and extraction instructions in a JSON schema via response_format; it does not follow ordinary chat or system prompts.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.025}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Inference.net's 3B-parameter HTML-to-JSON extraction model, optimized for throughput and low cost on high-volume workloads. It turns HTML into typed, structured data for web scraping and product catalog ingestion, with a 128K-token context window. Supply HTML in the user message and extraction instructions in a JSON schema via response_format; it does not follow ordinary chat or system prompts.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.015}},"thinkingmachines/inkling:thinking":{"id":"thinkingmachines/inkling:thinking","name":"Inkling Thinking","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/Inkling-Small:thinking":{"id":"thinkingmachines/Inkling-Small:thinking","name":"Inkling Small Thinking","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Data Used for Training)","description":"A much cheaper opt-in version of Muse Spark 1.2 with the same multimodal coding and agentic capabilities. Prompts and outputs sent to this Contributor model may be used by Meta for training and to improve its products; use the standard Muse Spark 1.2 model if you do not want your data used for training.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Meta's Muse Spark 1.3 Contributor is a frontier multimodal reasoning model for long-horizon coding and agentic workflows, with strong gains in computer use, browsing, professional tool use, codebase understanding, and million-token retrieval. It accepts text, images, audio, video, and files, supports tool calling and structured output, and always reasons before answering. Prompts and outputs may be used by Meta for training and to improve its products.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"mlabonne/NeuralDaredevil-8B-abliterated":{"id":"mlabonne/NeuralDaredevil-8B-abliterated","name":"Neural Daredevil 8B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.44,"output":0.44,"cache_read":0.22}},"NeverSleep/Lumimaid-v0.2-70B":{"id":"NeverSleep/Lumimaid-v0.2-70B","name":"Lumimaid v0.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1,"output":1.5,"cache_read":0.5}},"nanogpt/coding-router:low":{"id":"nanogpt/coding-router:low","name":"Coding Router Low","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"nanogpt/coding-router:medium":{"id":"nanogpt/coding-router:medium","name":"Coding Router Medium","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"nanogpt/coding-router":{"id":"nanogpt/coding-router","name":"Coding Router","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"nanogpt/coding-router:max":{"id":"nanogpt/coding-router:max","name":"Coding Router Max","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"nanogpt/coding-router:high":{"id":"nanogpt/coding-router:high","name":"Coding Router High","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"LatitudeGames/Wayfarer-Large-70B-Llama-3.3":{"id":"LatitudeGames/Wayfarer-Large-70B-Llama-3.3","name":"Llama 3.3 70B Wayfarer","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"VongolaChouko/Starcannon-Unleashed-12B-v1.0":{"id":"VongolaChouko/Starcannon-Unleashed-12B-v1.0","name":"Mistral Nemo Starcannon 12b v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":80000},"cost":{"input":0.25,"output":0.9,"cache_read":0.125}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-latest":{"id":"x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":500000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B":{"id":"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B","name":"Nemotron Tenyxchat Storybreaker 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B":{"id":"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B","name":"Llama 3.05 Storybreaker Ministral 70b","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"upstage/solar-mini4:thinking":{"id":"upstage/solar-mini4:thinking","name":"Solar Mini 4 Thinking","description":"Solar Mini 4 with reasoning enabled for agentic tasks and harder analysis across a 524K-token context window.","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.005}},"upstage/solar-pro4:thinking":{"id":"upstage/solar-pro4:thinking","name":"Solar Pro 4 Thinking","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"upstage/solar-mini4":{"id":"upstage/solar-mini4","name":"Solar Mini 4","description":"Upstage's compact 35B-parameter mixture-of-experts model with 3B active parameters and a 524K context window. Built for fast, cost-efficient agentic tasks, with strong Korean and English support.","family":"solar","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.005}},"bytedance/doubao-seed-2.1-turbo":{"id":"bytedance/doubao-seed-2.1-turbo","name":"Doubao Seed 2.1 Turbo","description":"Fast, lower-cost model in the Doubao Seed 2.1 family for everyday chat, coding assistance, document work, and high-throughput productivity tasks. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance/doubao-seed-2.1-pro":{"id":"bytedance/doubao-seed-2.1-pro","name":"Doubao Seed 2.1 Pro","description":"Higher-capability model in the Doubao Seed 2.1 family for agentic coding, long-context analysis, complex instruction following, and productivity workflows. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":1,"output":5,"cache_read":0.5}},"bytedance/doubao-seed-character":{"id":"bytedance/doubao-seed-character","name":"Doubao Seed Character","description":"ByteDance's character-focused Doubao Seed model for roleplay, persona consistency, dialogue, and creative character interactions. It supports text and image input with a 128k context window. Requests route through ZenMux to ByteDance; ZenMux does not publish a model-API zero-retention or training guarantee, so avoid sensitive data.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"release_date":"2026-07-18","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1179,"output":0.2947,"cache_read":0.0236,"cache_write":0.0025}},"liquid/lfm-2.5-2.6b":{"id":"liquid/lfm-2.5-2.6b","name":"LFM2.5 2.6B","description":"Liquid AI's compact 2.6B reasoning model for agent workflows, data extraction, RAG, and long-context processing. It supports tool calling and structured output, but Liquid advises against using it for agentic coding. Warning: prompts and responses may be logged and used for model training or service improvement; do not send sensitive data.","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"google/gemini-3.1-pro-preview-low":{"id":"google/gemini-3.1-pro-preview-low","name":"Gemini 3.1 Pro (Preview Low)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.5-flash-thinking":{"id":"google/gemini-3.5-flash-thinking","name":"Gemini 3.5 Flash Thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemma-4-26b-a4b-it:thinking":{"id":"google/gemma-4-26b-a4b-it:thinking","name":"Gemma 4 26B A4B Thinking","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.13,"output":0.4,"cache_read":0.065}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/diffusiongemma":{"id":"google/diffusiongemma","name":"DiffusionGemma","description":"DiffusionGemma is a high-speed diffusion-based version of Gemma 4 26B A4B. It supports optional reasoning and a 262,144-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","xhigh"]}],"tool_call":false,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemma-4-31b-it:thinking":{"id":"google/gemma-4-31b-it:thinking","name":"Gemma 4 31B Thinking","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.35,"cache_read":0.05}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview-high":{"id":"google/gemini-3.1-pro-preview-high","name":"Gemini 3.1 Pro (Preview High)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemma4-31b-splituntied":{"id":"google/gemma4-31b-splituntied","name":"Gemma 4 31B Split-Untied","description":"Blazed-Forge's Split-Untied is a text-only Gemma 4 31B community finetune with an untied BF16 output head, built for creative writing, roleplay, expressive dialogue, and tool use.","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemma-4-26b-a4b-it-cybersecurity":{"id":"google/gemma-4-26b-a4b-it-cybersecurity","name":"Gemma 4 26B A4B Cybersecurity","description":"Gemma 4 26B A4B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1056,"output":0.3344,"cache_read":0.0528}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro (Preview Custom Tools)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3-flash-preview-thinking":{"id":"google/gemini-3-flash-preview-thinking","name":"Gemini 3 Flash Thinking","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"TEE/gemma-4-31b-it":{"id":"TEE/gemma-4-31b-it","name":"Gemma 4 31B IT TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.15,"output":0.46,"cache_read":0.075}},"TEE/qwen3.8-27b":{"id":"TEE/qwen3.8-27b","name":"Qwen3.8 27B TEE","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"TEE/nemotron-3.5-lightning":{"id":"TEE/nemotron-3.5-lightning","name":"Nvidia Nemotron 3.5 Lightning TEE","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.08,"output":0.2,"cache_read":0.04}},"TEE/glm-5.3-flash":{"id":"TEE/glm-5.3-flash","name":"GLM 5.3 Flash TEE","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"TEE/qwen3.5-27b":{"id":"TEE/qwen3.5-27b","name":"Qwen3.5 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"TEE/muse-glimmer-30b":{"id":"TEE/muse-glimmer-30b","name":"Muse Glimmer 30B TEE","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"TEE/kimi-k3":{"id":"TEE/kimi-k3","name":"Kimi K3 TEE","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":1.5}},"TEE/deepseek-v4.1-flash":{"id":"TEE/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash TEE","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.65,"output":1.45,"cache_read":0.13}},"TEE/llama3-3-70b":{"id":"TEE/llama3-3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1.75,"output":2.75,"cache_read":1.75}},"TEE/kimi-k2.6":{"id":"TEE/kimi-k2.6","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.5,"output":5.25,"cache_read":0.375}},"TEE/gemma-4-26b-a4b-uncensored":{"id":"TEE/gemma-4-26b-a4b-uncensored","name":"Gemma 4 26B A4B Uncensored TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-23","last_updated":"2026-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":65536},"cost":{"input":0.15,"output":0.7,"cache_read":0.075}},"TEE/gemma4-31b:thinking":{"id":"TEE/gemma4-31b:thinking","name":"Gemma 4 31B Thinking TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-02","last_updated":"2026-05-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/gemma4-31b":{"id":"TEE/gemma4-31b","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-04","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/qwen3.6-35b-a3b":{"id":"TEE/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B TEE","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.2,"output":1.27,"cache_read":0.1}},"TEE/glm-5.2:thinking":{"id":"TEE/glm-5.2:thinking","name":"GLM 5.2 Thinking TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/qwen3.5-397b-a17b":{"id":"TEE/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.55,"output":3.5,"cache_read":0.275}},"TEE/glm-5.2":{"id":"TEE/glm-5.2","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/glm-5.1-thinking":{"id":"TEE/glm-5.1-thinking","name":"GLM 5.1 Thinking TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/glm-5.1":{"id":"TEE/glm-5.1","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/qwen2.5-vl-72b-instruct":{"id":"TEE/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"TEE/qwen3.6-27b":{"id":"TEE/qwen3.6-27b","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.32,"output":2.7,"cache_read":0.16}},"TEE/deepseek-v3.2":{"id":"TEE/deepseek-v3.2","name":"DeepSeek V3.2 TEE","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"input":164000,"output":65536},"cost":{"input":0.5,"output":1,"cache_read":0.25}},"TEE/gpt-oss-120b":{"id":"TEE/gpt-oss-120b","name":"GPT-OSS 120B TEE","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":2,"output":2,"cache_read":2}},"TEE/glm-5.3":{"id":"TEE/glm-5.3","name":"GLM 5.3 TEE","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"TEE/kimi-k2.7-code":{"id":"TEE/kimi-k2.7-code","name":"Kimi K2.7 Code TEE","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2","name":"EVA-Qwen2.5-72B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0","name":"EVA Llama 3.33 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2","name":"EVA-Qwen2.5-32B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1","name":"EVA-LLaMA-3.33-70B-v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16":{"id":"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16","name":"Llama 3.1 70B Celeste v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"MarinaraSpaghetti/NemoMix-Unleashed-12B":{"id":"MarinaraSpaghetti/NemoMix-Unleashed-12B","name":"NemoMix 12B Unleashed","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"pamanseau/OpenReasoning-Nemotron-32B":{"id":"pamanseau/OpenReasoning-Nemotron-32B","name":"OpenReasoning Nemotron 32B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated","name":"DeepSeek R1 Qwen Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1.4,"output":1.4,"cache_read":0.7}},"huihui-ai/Qwen2.5-32B-Instruct-abliterated":{"id":"huihui-ai/Qwen2.5-32B-Instruct-abliterated","name":"Qwen 2.5 32B Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-06","last_updated":"2025-01-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/Llama-3.3-70B-Instruct-abliterated":{"id":"huihui-ai/Llama-3.3-70B-Instruct-abliterated","name":"Llama 3.3 70B Instruct abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated","name":"DeepSeek R1 Llama 70B Abliterated","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"Gryphe/MythoMax-L2-13b":{"id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"input":4096,"output":3686},"cost":{"input":0.1003,"output":0.1003,"cache_read":0.05015}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"IBM Granite 4.2 8B is an Apache 2.0-licensed dense model with native step-by-step reasoning and specialized training for agentic work. It can plan before acting, sequence tools, navigate codebases, work in terminals, and verify results across coding, search, mathematics, science, and complex instruction-following tasks.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"TheDrummer/skyfall-36b-v2":{"id":"TheDrummer/skyfall-36b-v2","name":"TheDrummer Skyfall 36B V2","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"TheDrummer/Artemis-v1.1":{"id":"TheDrummer/Artemis-v1.1","name":"TheDrummer/Artemis v1.1","description":"TheDrummer's Artemis v1.1 is a Gemma 4 31B fine-tune for creative writing, expressive dialogue, and roleplay, with optional thinking and a 262K context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-06","last_updated":"2026-09-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"TheDrummer/Anubis-70B-v1":{"id":"TheDrummer/Anubis-70B-v1","name":"Anubis 70B v1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Anubis-70B-v1.1":{"id":"TheDrummer/Anubis-70B-v1.1","name":"Anubis 70B v1.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Cydonia-24B-v4.3":{"id":"TheDrummer/Cydonia-24B-v4.3","name":"The Drummer Cydonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.12,"output":0.15,"cache_read":0.06}},"TheDrummer/Magidonia-24B-v4.3":{"id":"TheDrummer/Magidonia-24B-v4.3","name":"The Drummer Magidonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Cydonia-24B-v4.1":{"id":"TheDrummer/Cydonia-24B-v4.1","name":"The Drummer Cydonia 24B v4.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":0.55,"cache_read":0.16}},"TheDrummer/UnslopNemo-12B-v4.1":{"id":"TheDrummer/UnslopNemo-12B-v4.1","name":"UnslopNemo 12b v4","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":26214},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"TheDrummer/Cydonia-24B-v2":{"id":"TheDrummer/Cydonia-24B-v2","name":"The Drummer Cydonia 24B v2","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Rocinante-12B-v1.1":{"id":"TheDrummer/Rocinante-12B-v1.1","name":"Rocinante 12b","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.408,"output":0.595,"cache_read":0.204}},"TheDrummer/Cydonia-24B-v4":{"id":"TheDrummer/Cydonia-24B-v4","name":"The Drummer Cydonia 24B v4","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.2006,"output":0.2414,"cache_read":0.1003}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/deepseek-v3.2-exp-thinking":{"id":"deepseek-ai/deepseek-v3.2-exp-thinking","name":"DeepSeek V3.2 Exp Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":32768},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"deepseek-ai/deepseek-v3.2-exp":{"id":"deepseek-ai/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-V3.1:thinking":{"id":"deepseek-ai/DeepSeek-V3.1:thinking","name":"DeepSeek V3.1 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.1-Terminus:thinking":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus:thinking","name":"DeepSeek V3.1 Terminus (Thinking)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"stepfun-ai/step-3.5-flash-2603":{"id":"stepfun-ai/step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"MiniMaxAI/MiniMax-M1-80k":{"id":"MiniMaxAI/MiniMax-M1-80k","name":"MiniMax M1 80K","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.6052,"output":2.4225,"cache_read":0.3026}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"mistralai/mistral-small-4-119b-2603:thinking":{"id":"mistralai/mistral-small-4-119b-2603:thinking","name":"Mistral Small 4 119B Thinking","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-medium-3.5":{"id":"mistralai/mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Mistral Medium 3.5 is a 128B dense open-weights flagship model for instruction-following, reasoning, coding, long-horizon agentic work, tool use, structured output, and multimodal prompts. It supports a 256k context window and configurable reasoning effort.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 24B","description":"Mistral Small 24B hosted by IONOS in Berlin, Germany. Zero data retention.","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.1155,"output":0.3465}},"mistralai/mixtral-8x22b-instruct-v0.1":{"id":"mistralai/mixtral-8x22b-instruct-v0.1","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":26214},"cost":{"input":0.1989,"output":0.595,"cache_read":0.09945}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral Small 3.2 24B (2506)","description":"The latest iteration of Mistral Small, version 3.2 (2506) brings enhanced performance and capabilities. With 24 billion parameters, this model delivers state-of-the-art results across text generation tasks with improved efficiency.","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.4,"cache_read":0.1}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":2.006,"output":6.001,"cache_read":0.2}},"mistralai/mistral-medium-3.5:thinking":{"id":"mistralai/mistral-medium-3.5:thinking","name":"Mistral Medium 3.5 Thinking","description":"Mistral Medium 3.5 with reasoning enabled by default (reasoning_effort=high), for complex coding, agentic, and multi-step reasoning prompts.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"mistralai/devstral-small-2505":{"id":"mistralai/devstral-small-2505","name":"Mistral Devstral Small 2505","description":"OpenHands+Devstral is 100% local 100% open, and is SOTA for the category on SWE-Bench Verified: 46.8% accuracy.","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.06,"output":0.06,"cache_read":0.03}},"mistralai/devstral-2-123b-instruct-2512":{"id":"mistralai/devstral-2-123b-instruct-2512","name":"Devstral 2 123B","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B (2503)","description":"Building upon Mistral Small 3 (2501), Mistral Small 3.1 (2503) adds state-of-the-art vision understanding and enhances long context capabilities up to 128k tokens without compromising text performance. With 24 billion parameters, this model achieves top-tier capabilities in both text and vision tasks.","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.05}},"mistralai/mistral-nemo-instruct-2407":{"id":"mistralai/mistral-nemo-instruct-2407","name":"Mistral Nemo","description":"12B parameter model with multilingual support.","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Sakana AI's cost-performance Fugu model uses learned multi-agent orchestration to route tasks across expert models for reasoning, coding, and tool use.","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-ultra-v1.1":{"id":"sakana/fugu-ultra-v1.1","name":"Fugu Ultra v1.1","description":"Sakana AI's upgraded Fugu Ultra release with stronger coding, agentic task execution, and advanced reasoning through dynamic orchestration of frontier models.","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5,"output":30,"cache_read":0.5}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Ling-3.0-flash is a 124B-parameter Mixture-of-Experts model with approximately 5.1B parameters active per token. It prioritizes token efficiency and production-scale agentic inference, helping coding and tool-using agents complete more work within constrained latency and serving budgets.","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL is inclusionAI's native multimodal Mixture-of-Experts model with 124B total parameters and 5.5B active parameters per token. It combines image and video understanding with reasoning and tool use for document analysis, charts, visual verification, and interface-based agent tasks. Thinking is enabled by default and can be turned off in settings.","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash:thinking":{"id":"inclusionai/ling-3.0-flash:thinking","name":"Ling 3.0 Flash Thinking","description":"Ling-3.0-flash Thinking enables visible reasoning on inclusionAI's token-efficient 124B-parameter Mixture-of-Experts model for harder coding, tool use, planning, and production-scale agent workflows.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High-Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.9,"output":8,"cache_read":0.32}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/kimi-k2.5:thinking":{"id":"moonshotai/kimi-k2.5:thinking","name":"Kimi K2.5 Thinking","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.6:thinking":{"id":"moonshotai/kimi-k2.6:thinking","name":"Kimi K2.6 Thinking","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/kimi-k2-instruct-0711":{"id":"moonshotai/kimi-k2-instruct-0711","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-latest":{"id":"moonshotai/kimi-latest","name":"Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":100352},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"THUDM/GLM-Z1-9B-0414":{"id":"THUDM/GLM-Z1-9B-0414","name":"GLM Z1 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-z","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-4-9B-0414":{"id":"THUDM/GLM-4-9B-0414","name":"GLM 4 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-4-32B-0414":{"id":"THUDM/GLM-4-32B-0414","name":"GLM 4 32B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nvidia Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nvidia Nemotron 3.5 Lightning","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1","name":"Nvidia Nemotron Super 49B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nvidia Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nvidia Nemotron 3 Ultra 550B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/nemotron-3-super-120b-a12b:thinking":{"id":"nvidia/nemotron-3-super-120b-a12b:thinking","name":"Nvidia Nemotron 3 Super 120B Thinking","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/nemotron-3-ultra-550b-a55b:thinking":{"id":"nvidia/nemotron-3-ultra-550b-a55b:thinking","name":"Nvidia Nemotron 3 Ultra 550B Thinking","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF":{"id":"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF","name":"Nvidia Nemotron 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"nvidia/nemotron-3.5-lightning:thinking":{"id":"nvidia/nemotron-3.5-lightning:thinking","name":"Nvidia Nemotron 3.5 Lightning Thinking","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"xiaomi/mimo-v2.5-pro:thinking":{"id":"xiaomi/mimo-v2.5-pro:thinking","name":"MiMo V2.5 Pro Thinking","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo V2.6 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.6-pro-ultraspeed":{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","name":"MiMo V2.6 Pro UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036,"cache_write":0}},"xiaomi/mimo-v2.5:thinking":{"id":"xiaomi/mimo-v2.5:thinking","name":"MiMo V2.5 Thinking","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo V2.6 Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"input":6144,"output":4096},"cost":{"input":0.799,"output":1.207,"cache_read":0.3995}},"Salesforce/Llama-xLAM-2-70b-fc-r":{"id":"Salesforce/Llama-xLAM-2-70b-fc-r","name":"Llama-xLAM-2 70B fc-r","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":2.5,"cache_read":1.25}},"minimax/minimax-latest":{"id":"minimax/minimax-latest","name":"MiniMax Latest","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"minimax/minimax-m2.7-turbo":{"id":"minimax/minimax-m2.7-turbo","name":"MiniMax M2.7 Turbo","description":"Efficient MiniMax model for quick assistance, coding, and routine automation","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.3}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax 01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"input":1000192,"output":16384},"cost":{"input":0.1394,"output":1.122,"cache_read":0.0697}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax M2-her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65532,"input":65532,"output":2048},"cost":{"input":0.302,"output":1.207,"cache_read":0.151}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.315,"output":1.26,"cache_read":0.1575}},"minimax/minimax-m3:thinking":{"id":"minimax/minimax-m3:thinking","name":"MiniMax M3 Thinking","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.165}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.17,"output":1.53,"cache_read":0.085}},"inflatebot/MN-12B-Mag-Mell-R1":{"id":"inflatebot/MN-12B-Mag-Mell-R1","name":"Mag Mell R1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"kitani/clover-1-150b":{"id":"kitani/clover-1-150b","name":"Clover 1 150B Preview","description":"Kitani's 150B multimodal reasoning model for coding, tool use, and everyday assistance. This is a preview release and may have reliability issues.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-24","last_updated":"2026-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":0.15,"output":0.8}},"soob3123/GrayLine-Qwen3-8B":{"id":"soob3123/GrayLine-Qwen3-8B","name":"Grayline Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/Veiled-Calla-12B":{"id":"soob3123/Veiled-Calla-12B","name":"Veiled Calla 12B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/amoral-gemma3-27B-v2":{"id":"soob3123/amoral-gemma3-27B-v2","name":"Amoral Gemma3 27B v2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-23","last_updated":"2025-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"stepfun/step-5-preview":{"id":"stepfun/step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"stepfun/step-3.7-flash:thinking":{"id":"stepfun/step-3.7-flash:thinking","name":"Step 3.7 Flash Thinking","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"Sao10K/L3.1-70B-Hanami-x1":{"id":"Sao10K/L3.1-70B-Hanami-x1","name":"Llama 3.1 70B Hanami","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Sao10K/L3.1-70B-Euryale-v2.2":{"id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Llama 3.1 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.306,"output":0.357,"cache_read":0.153}},"Sao10K/L3.3-70B-Euryale-v2.3":{"id":"Sao10K/L3.3-70B-Euryale-v2.3","name":"Llama 3.3 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Sao10K/L3-8B-Stheno-v3.2":{"id":"Sao10K/L3-8B-Stheno-v3.2","name":"Sao10K Stheno 8b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"inception/mercury-2.5-preview":{"id":"inception/mercury-2.5-preview","name":"Mercury 2.5 Preview","description":"Mercury 2.5 Preview is Inception's latest and most intelligent diffusion language model. Instead of generating tokens strictly one at a time, it produces and refines multiple tokens in parallel, reaching up to 1,107 tokens per second on standard GPUs. It delivers a 10+ point intelligence gain over Mercury 2, with tunable reasoning, parallel tool calls, schema-aligned JSON output, and a 260K context window. It is built for latency-sensitive production work such as search agents, voice pipelines, customer support, rapid coding iteration, and coding subagents.","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"input":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"ornith-ai/ornith-1.5-35b-a3b":{"id":"ornith-ai/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, tool use, image understanding, and long-context work. This variant disables thinking for faster direct responses.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"ornith-ai/ornith-1.5-35b-a3b:thinking":{"id":"ornith-ai/ornith-1.5-35b-a3b:thinking","name":"Ornith 1.5 35B Thinking","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, reasoning, tool use, image understanding, and long-context work. This variant enables thinking by default.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":5120},"cost":{"input":0.0595,"output":0.238,"cache_read":0.02975}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon Nova 2 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65535},"cost":{"input":0.51,"output":4.25,"cache_read":0.255}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":32000},"cost":{"input":0.799,"output":3.196,"cache_read":0.3995}},"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond":{"id":"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond","name":"MS3.2 24B Magnum Diamond","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Llama 3.1 8b (uncensored)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.8,"output":1.6,"cache_read":0.4}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion 3.0","description":"Aion 3.0 is a GLM-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5":{"id":"aion-labs/aion-3.5","name":"AionLabs: Aion 3.5","description":"A GLM-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5-mini":{"id":"aion-labs/aion-3.5-mini","name":"AionLabs: Aion 3.5 Mini","description":"A GLM-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion 3.0 Mini","description":"Aion 3.0 Mini is a DeepSeek-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3.8-max:thinking":{"id":"qwen/qwen3.8-max:thinking","name":"Qwen3.8 Max Thinking","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":6,"cache_read":0.25}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"qwen/qwen3.5-omni-plus":{"id":"qwen/qwen3.5-omni-plus","name":"Qwen3.5 Omni Plus","description":"Qwen3.5 Omni Plus is Qwen's stronger general multimodal model. We verified live support for text prompts, images, audio files, and direct video URLs on Alibaba's chat-completions-compatible API. Alibaba describes Plus as a comprehensive evolution of Qwen3 Omni with support for over 10 hours of audio input.","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536}},"qwen/qwen3.7-flash:thinking":{"id":"qwen/qwen3.7-flash:thinking","name":"Qwen3.7 Flash Thinking","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen 3 235b A22B 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen 2.5 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":1.5997,"output":6.392,"cache_read":0.79985}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.2,"output":1.5,"cache_read":0.1}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"input":262000,"output":65536},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/qwen3.5-35b-a3b:thinking":{"id":"qwen/qwen3.5-35b-a3b:thinking","name":"Qwen3.5 35B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen 3 32b","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":995904,"input":995904,"output":32768},"cost":{"input":0.3995,"output":1.2002,"cache_read":0.19975}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"qwen/qwen3.8-27b-obliterated:thinking":{"id":"qwen/qwen3.8-27b-obliterated:thinking","name":"Qwen 3.8 27B Obliterated Thinking","description":"Qwen 3.8 27B Obliterated with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.2}},"qwen/qwen-turbo":{"id":"qwen/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":8192},"cost":{"input":0.04998,"output":0.2006,"cache_read":0.02499}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3.8-27b:thinking":{"id":"qwen/qwen3.8-27b:thinking","name":"Qwen3.8 27B Thinking","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen/qwen3.8-27b-obliterated":{"id":"qwen/qwen3.8-27b-obliterated","name":"Qwen 3.8 27B Obliterated","description":"Qwen 3.8 27B Obliterated is an open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.2}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"qwen/qwen3.8-27b-uncensored":{"id":"qwen/qwen3.8-27b-uncensored","name":"Qwen 3.8 27B Uncensored","description":"Qwen 3.8 27B Uncensored is an NVFP4 open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.15,"output":1.2,"cache_read":0.125}},"qwen/qwen3.7-max:thinking":{"id":"qwen/qwen3.7-max:thinking","name":"Qwen3.7 Max Thinking","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen 3 235b A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"qwen/qwen3.8-27b-uncensored:thinking":{"id":"qwen/qwen3.8-27b-uncensored:thinking","name":"Qwen 3.8 27B Uncensored Thinking","description":"Qwen 3.8 27B Uncensored with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.15,"output":1.2,"cache_read":0.125}},"qwen/qwen3.5-omni-flash":{"id":"qwen/qwen3.5-omni-flash","name":"Qwen3.5 Omni Flash","description":"Qwen3.5 Omni Flash is Qwen's fast multimodal model. We verified live support for text prompts, images, audio files, and direct video URLs on Alibaba's chat-completions-compatible API. Alibaba describes Flash as a fully evolved version of Qwen3 Omni with audio input support across 60+ languages.","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":49152,"input":49152,"output":16384}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B (Instruct)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3-max-2026-01-23":{"id":"qwen/qwen3-max-2026-01-23","name":"Qwen3 Max 2026-01-23","description":"Qwen3 Max is Alibaba's flagship Qwen 3 reasoning model with native tool use (web search, web extractor, code interpreter) and a 256K context window.","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-01-26","last_updated":"2026-01-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"qwen/qwen3.8-27b-fable":{"id":"qwen/qwen3.8-27b-fable","name":"Qwen 3.8 27B Fable","description":"Qwen 3.8 27B Fable is an open-weight multimodal creative finetune for expressive dialogue, long-form storytelling, character work, and roleplay.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.8-27b-hemmingway":{"id":"qwen/qwen3.8-27b-hemmingway","name":"Qwen 3.8 27B Hemingway","description":"Qwen 3.8 27B Hemingway is an open-weight NVFP4 multimodal creative finetune for long-form prose, character dialogue, storytelling, and roleplay.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.19,"output":1.16,"cache_read":0.02,"cache_write":0.24}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"qwen/qwen3.5-plus:thinking":{"id":"qwen/qwen3.5-plus:thinking","name":"Qwen3.5 Plus Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.5}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"qwen/qwen3.5-122b-a10b:thinking":{"id":"qwen/qwen3.5-122b-a10b:thinking","name":"Qwen3.5 122B A10B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"qwen/qwen3.5-397b-a17b:thinking":{"id":"qwen/qwen3.5-397b-a17b:thinking","name":"Qwen3.5 397B A17B Thinking","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/qwen3.8-max-prime":{"id":"qwen/qwen3.8-max-prime","name":"Qwen3.8 Max Prime","description":"High-throughput edition of Qwen3.8 Max for coding, professional work, multimodal understanding, and long-running agent workflows","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":4,"output":12,"cache_read":0.5}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":0.14,"output":0.42,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-27b-cybersecurity":{"id":"qwen/qwen3.8-27b-cybersecurity","name":"Qwen 3.8 27B Cybersecurity","description":"Qwen 3.8 27B Cybersecurity is a cybersecurity-focused variant based on the uncensored model, with provider moderation for illegal activities. It supports optional reasoning, image understanding, tool calling, and a 262,144-token context window.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-19","last_updated":"2026-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.6,"cache_read":0.05}},"qwen/qwen-long":{"id":"qwen/qwen-long","name":"Qwen Long 10M","description":"Alibaba's huge context window model. Takes in up to 10 million tokens, which is equivalent to dozens of books.","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-08-01","last_updated":"2024-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":8192},"cost":{"input":0.1003,"output":0.408,"cache_read":0.05015}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245760,"input":245760,"output":65536},"cost":{"input":1.04,"output":6.24,"cache_read":0.52}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen 3 8B","description":"Qwen 3 8B is a 8B model. Supports switching between thinking and non thinking: trigger thinking with /think and /no_think anywhere in a prompt or system message to toggle chain-of-thought reasoning.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.47,"output":0.47,"cache_read":0.235}},"qwen/qwen3.5-27b:thinking":{"id":"qwen/qwen3.5-27b:thinking","name":"Qwen3.5 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"qwen/qwen3.8-27b-queen":{"id":"qwen/qwen3.8-27b-queen","name":"Qwen 3.8 27B Queen","description":"Qwen 3.8 27B Queen is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 262,144-token context window.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B-parameter mixture-of-experts language model from Qwen, with 3.3B active parameters per inference. Significant improvements in general capabilities, including instruction following, logical reasoning, text comprehension, mathematics, science, coding and tool usage.","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"qwen/qwen2.5-coder-32b-instruct":{"id":"qwen/qwen2.5-coder-32b-instruct","name":"Qwen 2.5 Coder 32b","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"qwen/qwen3.7-plus:thinking":{"id":"qwen/qwen3.7-plus:thinking","name":"Qwen3.7 Plus Thinking","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"qwen/qwen3.6-35b-a3b:thinking":{"id":"qwen/qwen3.6-35b-a3b:thinking","name":"Qwen3.6 35B A3B Thinking","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen 3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_read":0.0325,"cache_write":0.40625}},"qwen/qwen3.6-27b:thinking":{"id":"qwen/qwen3.6-27b:thinking","name":"Qwen3.6 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen 3 14b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.08,"output":0.24,"cache_read":0.04}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3.5-flash:thinking":{"id":"qwen/qwen3.5-flash:thinking","name":"Qwen3.5 Flash Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.17,"cache_write":2.5}},"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5":{"id":"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5","name":"Llama 3 70B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-terra-latest":{"id":"openai/gpt-terra-latest","name":"GPT Terra Latest","description":"Compatibility alias that routes to GPT 5.6 Terra, the latest supported GPT Terra model.","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT 6 Astra Pro","description":"GPT 6 Astra in Pro reasoning mode. Uses additional model work for difficult tasks, with higher latency and token usage at the same per-token rates. Reasoning effort remains independently configurable.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/o3-pro-2025-06-10":{"id":"openai/o3-pro-2025-06-10","name":"OpenAI o3-pro (2025-06-10)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":22,"output":88,"cache_read":11}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT 5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT 5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT 5.6 Sol Pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT 5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o4-mini":{"id":"openai/o4-mini","name":"OpenAI o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3-mini":{"id":"openai/o3-mini","name":"OpenAI o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT 4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT 5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT 5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":15,"output":120,"cache_read":1.5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT 6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-luna-latest":{"id":"openai/gpt-luna-latest","name":"GPT Luna Latest","description":"Compatibility alias that routes to GPT 5.6 Luna, the latest supported GPT Luna model.","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT 5.6 Luna Pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT 5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.075,"output":0.3}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI o3-mini (High)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-sol-latest":{"id":"openai/gpt-sol-latest","name":"GPT Sol Latest","description":"Compatibility alias that routes to GPT 5.6 Sol, the latest supported GPT Sol model.","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-6-sol-pro":{"id":"openai/gpt-6-sol-pro","name":"GPT 6 Sol Pro","description":"GPT-6 Sol Pro uses the same underlying model as GPT-6 Sol with Pro reasoning mode enabled for higher-quality responses on complex tasks.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-astra-latest":{"id":"openai/gpt-astra-latest","name":"GPT Astra Latest","description":"Compatibility alias that routes to GPT 6 Astra, the latest supported GPT Astra model.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT 5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT 4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/o3-mini-low":{"id":"openai/o3-mini-low","name":"OpenAI o3-mini (Low)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-01-31","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT 4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT 5.6 Terra Pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT 6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI o4-mini high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-12-04","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-6-luna-pro":{"id":"openai/gpt-6-luna-pro","name":"GPT 6 Luna Pro","description":"GPT-6 Luna Pro uses the same underlying model as GPT-6 Luna with Pro reasoning mode enabled for higher-quality responses on complex tasks.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.35,"output":0.75}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT 5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o3":{"id":"openai/o3","name":"OpenAI o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":1}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT 5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.1-2025-11-13":{"id":"openai/gpt-5.1-2025-11-13","name":"GPT-5.1 (2025-11-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT 6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/o1-pro":{"id":"openai/o1-pro","name":"OpenAI o1 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":150,"output":600,"cache_read":75}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"NousResearch/hermes-3-llama-3.1-70b":{"id":"NousResearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-01-07","last_updated":"2026-01-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.408,"output":0.408,"cache_read":0.204}},"NousResearch/hermes-4-405b":{"id":"NousResearch/hermes-4-405b","name":"Hermes 4 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"NousResearch/hermes-4-405b:thinking":{"id":"NousResearch/hermes-4-405b:thinking","name":"Hermes 4 Large (Thinking)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0":{"id":"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0","name":"Omega Directive 24B Unslop v2.0","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}}}},"abliteration-ai":{"id":"abliteration-ai","env":["ABLIT_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.abliteration.ai/v1","name":"abliteration.ai","doc":"https://docs.abliteration.ai/models","models":{"abliterated-model-large-v2":{"id":"abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"GLM-5.3 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliterated-model":{"id":"abliterated-model","name":"Abliterated Model","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-06","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":150000,"input":150000,"output":8192},"cost":{"input":3,"output":3,"cache_read":0.3}},"abliterated-model-large":{"id":"abliterated-model-large","name":"Abliterated Model Large","description":"GLM-5.2 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-25","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}}}},"crof":{"id":"crof","env":["CROF_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://crof.ai/v1","name":"CrofAI","doc":"https://crof.ai/docs","models":{"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.03}},"kimi-k3-eco":{"id":"kimi-k3-eco","name":"Kimi K3 Eco","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":1,"output":4,"cache_read":0.1}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.07,"output":0.22,"cache_read":0.01}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash (New)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.1,"cache_read":0.003}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":8,"cache_read":0.25}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.2,"cache_read":0.007}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.5,"output":1.99,"cache_read":0.05}},"greg-2-super":{"id":"greg-2-super","name":"Greg 2 Super","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":1.5,"output":5,"cache_read":0.25}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro (0813)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.01}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":0.8,"cache_read":0.003,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"greg-rp":{"id":"greg-rp","name":"Greg (Roleplay)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.04,"output":0.15,"cache_read":0.008}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":1.75,"cache_read":0.07}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.3,"output":1.05,"cache_read":0.05}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.45,"output":2.15,"cache_read":0.08,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.003}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.04}},"greg-2-ultra":{"id":"greg-2-ultra","name":"Greg 2 Ultra","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":3,"output":10,"cache_read":0.5}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.18,"output":0.35,"cache_read":0.04}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":1.4,"cache_read":0.06}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.55,"output":2.25,"cache_read":0.05}},"greg-1-mini":{"id":"greg-1-mini","name":"Greg 1 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.07,"output":0.15,"cache_read":0.01}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.12,"output":0.21,"cache_read":0.003}}}},"standardcompute":{"id":"standardcompute","env":["STANDARDCOMPUTE_API_KEY"],"npm":"@openrouter/ai-sdk-provider","api":"https://api.stdcmpt.com/v1","name":"Standard Compute","doc":"https://standardcompute.com/models","models":{"standardcompute":{"id":"standardcompute","name":"Standard Compute","description":"Flat-rate smart-routing gateway: one model id, each request routed across a curated catalog of 1M-context models (DeepSeek, GLM, MiniMax, Qwen, GPT-5.6, Claude 5, Gemini 2.5, Kimi) or pinned to a user-selected model","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":24576},"cost":{"input":0,"output":0}}}},"cloudferro-sherlock":{"id":"cloudferro-sherlock","env":["CLOUDFERRO_SHERLOCK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-sherlock.cloudferro.com/openai/v1/","name":"CloudFerro Sherlock","doc":"https://docs.sherlock.cloudferro.com/","models":{"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10-09","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":70000,"output":70000},"cost":{"input":2.92,"output":2.92}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"input":180000,"output":16000},"cost":{"input":0.3,"output":1.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":2.92,"output":2.92}},"speakleash/Bielik-11B-v2.6-Instruct":{"id":"speakleash/Bielik-11B-v2.6-Instruct","name":"Bielik 11B v2.6 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}},"speakleash/Bielik-11B-v3.0-Instruct":{"id":"speakleash/Bielik-11B-v3.0-Instruct","name":"Bielik 11B v3.0 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}}}},"anthropic":{"id":"anthropic","env":["ANTHROPIC_API_KEY"],"npm":"@ai-sdk/anthropic","name":"Anthropic","doc":"https://docs.anthropic.com/en/docs/about-claude/models","models":{"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.4,"cache_write":10},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-07","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-04","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-14","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}}}},"tinfoil":{"id":"tinfoil","env":["TINFOIL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.tinfoil.sh/v1","name":"Tinfoil","doc":"https://docs.tinfoil.sh","models":{"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"gpt-oss-safeguard-120b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":1.25,"cache_read":0.1}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":4,"output":20,"cache_read":0.8}},"llama3-3-70b":{"id":"llama3-3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":1.75,"output":2.75}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":1}},"glm-5-3":{"id":"glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.8,"output":5.75,"cache_read":0.45}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":0.65,"output":1.45,"cache_read":0.13}},"nomic-embed-text":{"id":"nomic-embed-text","name":"Nomic Embed Text v1.5","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2024-02","last_updated":"2024-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":768},"cost":{"input":0.05,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}}}},"llama":{"id":"llama","env":["LLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llama.com/compat/v1/","name":"Llama","doc":"https://llama.developer.meta.com/docs/models","models":{"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"groq-llama-4-maverick-17b-128e-instruct":{"id":"groq-llama-4-maverick-17b-128e-instruct","name":"Groq-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-4-scout-17b-16e-instruct-fp8":{"id":"llama-4-scout-17b-16e-instruct-fp8","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"cerebras-llama-4-maverick-17b-128e-instruct":{"id":"cerebras-llama-4-maverick-17b-128e-instruct","name":"Cerebras-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-8b-instruct":{"id":"llama-3.3-8b-instruct","name":"Llama-3.3-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"cerebras-llama-4-scout-17b-16e-instruct":{"id":"cerebras-llama-4-scout-17b-16e-instruct","name":"Cerebras-Llama-4-Scout-17B-16E-Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}}}},"cohere":{"id":"cohere","env":["COHERE_API_KEY"],"npm":"@ai-sdk/cohere","name":"Cohere","doc":"https://docs.cohere.com/docs/models","models":{"command-r-08-2024":{"id":"command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"north-mini-code-1-0":{"id":"north-mini-code-1-0","name":"North Mini Code","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.cohere.ai/compatibility/v1"},"cost":{"input":0,"output":0}},"command-a-translate-08-2025":{"id":"command-a-translate-08-2025","name":"Command A Translate","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":2.5,"output":10}},"c4ai-aya-expanse-8b":{"id":"c4ai-aya-expanse-8b","name":"Aya Expanse 8B","description":"Compact open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":4000}},"command-r7b-arabic-02-2025":{"id":"command-r7b-arabic-02-2025","name":"Command R7B Arabic","description":"Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"command-a-plus-05-2026":{"id":"command-a-plus-05-2026","name":"Command A Plus","description":"Cohere's stronger command model for multilingual agents and enterprise workflows","family":"command-a","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04-01","release_date":"2026-05-20","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":2.5,"output":10}},"command-a-03-2025":{"id":"command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"c4ai-aya-vision-8b":{"id":"c4ai-aya-vision-8b","name":"Aya Vision 8B","description":"Compact open multilingual vision model for OCR and visual question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"c4ai-aya-vision-32b":{"id":"c4ai-aya-vision-32b","name":"Aya Vision 32B","description":"Open multilingual vision model for OCR, visual reasoning, and image question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"command-r7b-12-2024":{"id":"command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"command-r-plus-08-2024":{"id":"command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"command-a-vision-07-2025":{"id":"command-a-vision-07-2025","name":"Command A Vision","description":"Cohere vision model for multilingual document analysis, OCR, and image understanding","family":"command-a","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":2.5,"output":10}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Command A Reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":2.5,"output":10}},"c4ai-aya-expanse-32b":{"id":"c4ai-aya-expanse-32b","name":"Aya Expanse 32B","description":"Open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}}}},"deepseek":{"id":"deepseek","env":["DEEPSEEK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.deepseek.com","name":"DeepSeek","doc":"https://api-docs.deepseek.com/quick_start/pricing","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"status":"deprecated","cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.435,"output":0.87,"reasoning":0.87,"cache_read":0.003625}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"status":"deprecated","cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}}}},"baseten":{"id":"baseten","env":["BASETEN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.baseten.co/v1","name":"Baseten","doc":"https://docs.baseten.co/inference/model-apis/overview","models":{"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1,"output":4.05}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":131000},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204000,"output":204000},"status":"deprecated","cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-30","last_updated":"2026-02-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.6,"output":3,"cache_read":0.12}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.3}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai-org/GLM-5.2-Fast":{"id":"zai-org/GLM-5.2-Fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai-org/GLM-5.3-Fast":{"id":"zai-org/GLM-5.3-Fast","name":"GLM 5.3 Fast","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.95,"output":3.15,"cache_read":0.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"nvidia/Nemotron-120B-A12B":{"id":"nvidia/Nemotron-120B-A12B","name":"Nemotron Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.3,"output":0.75,"cache_read":0.06}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128072,"output":128072},"cost":{"input":0.1,"output":0.5}}}},"nan":{"id":"nan","env":["NAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nan.builders/v1","name":"NaN","doc":"https://nan.builders/docs/models","models":{"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"glm5.3-flash":{"id":"glm5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"gemma4":{"id":"gemma4","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6":{"id":"qwen3.6","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"glm5.3":{"id":"glm5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}}}},"stepfun-ai-step-plan":{"id":"stepfun-ai-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/step_plan/v1","name":"StepFun Step Plan (Global)","doc":"https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api","models":{"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}}}},"nearai":{"id":"nearai","env":["NEARAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://cloud-api.near.ai/v1","name":"NEAR AI Cloud","doc":"https://docs.near.ai/","models":{"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"black-forest-labs/FLUX.2-klein-4B":{"id":"black-forest-labs/FLUX.2-klein-4B","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":1,"output":1}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"Qwen/Qwen3-Reranker-0.6B":{"id":"Qwen/Qwen3-Reranker-0.6B","name":"Qwen3 Reranker 0.6B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen3-VL 30B-A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.15,"output":0.55}},"Qwen/Qwen3-Embedding-0.6B":{"id":"Qwen/Qwen3-Embedding-0.6B","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen 3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.17,"output":1.1,"cache_read":0.056}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.4,"output":4.4}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.01,"output":0.01}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}}}},"wandb":{"id":"wandb","env":["WANDB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.wandb.ai/v1","name":"CoreWeave","doc":"https://docs.wandb.ai/inference","models":{"JetBrains/Mellum2-12B-A2.5B-Instruct":{"id":"JetBrains/Mellum2-12B-A2.5B-Instruct","name":"Mellum2 12B A2.5B","description":"Mellum2-12B-A2.5B-Instruct is a fast MoE model with 131K context built for coding, tool use, and low-latency AI workflows.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}},"meta-llama/Llama-3.1-70B-Instruct":{"id":"meta-llama/Llama-3.1-70B-Instruct","name":"Llama 3.1 70B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.8,"output":0.8,"cache_read":0.8}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.22,"output":0.22,"cache_read":0.22}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B","description":"Multilingual model excelling in conversational tasks, detailed instruction-following, and coding.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.71,"output":0.71,"cache_read":0.71}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B","description":"Gemma 4 31B Dense is designed for advanced reasoning, agentic workflows, and longer context and is natively trained on 140+ languages.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.34,"cache_read":0.1}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B","description":"Gemma 4 26B A4B is a multimodal MoE model with LoRA support and function calling for agentic workflows.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen3.8-27B is a dense multimodal model suited for coding, research, vision, and long-running agent tasks.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B MoE instruction-tuned model with enhanced reasoning, coding, and long-context understanding.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3,"cache_read":0.1}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen3.6-35B-A3B is an MoE multimodal model with 262K context optimized for agentic coding workflows.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5-35B-A3B","description":"Qwen3.5-35B-A3B is an open-weights multimodal MoE model built for efficient, high-throughput inference across chat, reasoning, and agentic tasks.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen3.6-27B is a 27B dense multimodal model with 262K context built for flagship-level agentic coding.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6,"cache_read":0.12}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Granite 4.2 8B is an instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-24","last_updated":"2026-08-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"ibm-granite/granite-4.1-8b":{"id":"ibm-granite/granite-4.1-8b","name":"Granite 4.1 8B","description":"Granite 4.1 8B is a long-context instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"A large hybrid model that supports both thinking and non-thinking modes via prompt templates.","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":161000,"output":161000},"cost":{"input":0.55,"output":1.65,"cache_read":0.55}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4-Flash-0731 is an MoE model great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4-Pro-0813 is a 1.6T-parameter MoE model excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.31,"output":3.96,"cache_read":0.044}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a multimodal MoE model for coding, reasoning, and agentic workloads with long contexts.","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.2,"output":0.65,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4-Flash is an MoE model with 1M context length great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4-Pro is a 1.6T-parameter MoE model with 49B active parameters excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.15,"output":2.55,"cache_read":0.2}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax M3","description":"MiniMax M3 is a multimodal MoE model with 23B active parameters optimized for coding and agentic workflows.","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.23,"output":0.96,"cache_read":0.05}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.65,"output":3.41,"cache_read":0.15}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi K2.7 Code is a 1T-parameter MoE model with 32B active parameters purpose-built for long-horizon agentic coding and software engineering.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":3.5,"cache_read":0.15}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"GLM-5.2 is a Mixture-of-Experts language model featuring 40 billion activated parameters and a total of 744 billion parameters.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.76,"output":2.42,"cache_read":0.14}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM-5.3-Flash is a natively multimodal model with 320B total parameters and 18B active parameters.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.05}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B","name":"Nemotron 3.5 Lightning","description":"Nemotron 3.5 Lightning is an MoE model built for fast, reliable agentic tasks across use cases such as financial services, cybersecurity, telecom, and retail.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.2,"cache_read":0.04}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron 3 Ultra","description":"Nemotron 3 Ultra is a powerful MoE model designed for long-running agents across coding, deep research, and enterprise automation.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.15,"cache_read":0.1}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"gpt-oss-20b","description":"Lower latency Mixture-of-Experts model trained on OpenAI's Harmony response format with reasoning capabilities.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.13,"cache_read":0.03}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Efficient Mixture-of-Experts model designed for high-reasoning, agentic and general-purpose use cases.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"OpenPipe/Qwen3-14B-Instruct":{"id":"OpenPipe/Qwen3-14B-Instruct","name":"Qwen3 14B Instruct","description":"An efficient multilingual, dense, instruction-tuned model, optimized by OpenPipe for building agents with finetuning.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.05,"output":0.22,"cache_read":0.05}}}},"subconscious":{"id":"subconscious","env":["SUBCONSCIOUS_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.subconscious.dev/v1","name":"Subconscious","doc":"https://docs.subconscious.dev","models":{"subconscious/tim-qwen3.6-27b":{"id":"subconscious/tim-qwen3.6-27b","name":"TIM-Qwen3.6 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":5000},"cost":{"input":0.3,"output":3,"cache_read":0.15}},"subconscious/glm-5.2":{"id":"subconscious/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"zeldoc":{"id":"zeldoc","env":["ZELDOC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.zeldoc.ai/v1","name":"Zeldoc","doc":"https://docs.zeldoc.ai","models":{"zdev":{"id":"zdev","name":"ZDev","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"databricks":{"id":"databricks","env":["DATABRICKS_HOST","DATABRICKS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1","name":"Databricks","doc":"https://docs.databricks.com/aws/en/machine-learning/foundation-models/","models":{"databricks-claude-opus-4-5":{"id":"databricks-claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-gpt-5-6-terra":{"id":"databricks-gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-kimi-k2-7-code":{"id":"databricks-kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"databricks-glm-5-2":{"id":"databricks-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"databricks-gemini-3-1-flash-lite":{"id":"databricks-gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"databricks-gpt-5-mini":{"id":"databricks-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"databricks-gemini-3-flash":{"id":"databricks-gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"databricks-claude-opus-4-7":{"id":"databricks-claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-gpt-oss-20b":{"id":"databricks-gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2}},"databricks-gpt-5-4-mini":{"id":"databricks-gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"databricks-gemini-3-pro":{"id":"databricks-gemini-3-pro","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-gpt-5-2":{"id":"databricks-gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"databricks-claude-haiku-4-5":{"id":"databricks-claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"databricks-gemini-3-1-pro":{"id":"databricks-gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-claude-opus-4-6":{"id":"databricks-claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-gpt-5-6-sol":{"id":"databricks-gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"databricks-claude-opus-4-1":{"id":"databricks-claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"databricks-gpt-oss-120b":{"id":"databricks-gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.072,"output":0.28}},"databricks-gemini-2-5-pro":{"id":"databricks-gemini-2-5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"databricks-gemini-2-5-flash":{"id":"databricks-gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"databricks-claude-sonnet-4-5":{"id":"databricks-claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-4":{"id":"databricks-gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-gpt-5-4-nano":{"id":"databricks-gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"databricks-claude-sonnet-4-6":{"id":"databricks-claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-6-luna":{"id":"databricks-gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"databricks-claude-sonnet-4":{"id":"databricks-claude-sonnet-4","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-1":{"id":"databricks-gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-5-nano":{"id":"databricks-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"databricks-gpt-5":{"id":"databricks-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-5-5":{"id":"databricks-gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"umans-ai-coding-plan":{"id":"umans-ai-coding-plan","env":["UMANS_AI_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI Coding Plan","doc":"https://app.umans.ai/offers/code/docs","models":{"umans-glm-5.3-flash":{"id":"umans-glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-qwen3.6-35b-a3b":{"id":"umans-qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"kuae-cloud-coding-plan":{"id":"kuae-cloud-coding-plan","env":["KUAE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-plan-endpoint.kuaecloud.net/v1","name":"KUAE Cloud Coding Plan","doc":"https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/","models":{"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"umans-ai":{"id":"umans-ai","env":["UMANS_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI","doc":"https://app.umans.ai/offers/code/docs/orgs","models":{"umans-glm-5.3-flash":{"id":"umans-glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131071},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}}}},"siliconflow":{"id":"siliconflow","env":["SILICONFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.com/v1","name":"SiliconFlow","doc":"https://cloud.siliconflow.com/models","models":{"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"nex-agi/Nex-N2-Pro":{"id":"nex-agi/Nex-N2-Pro","name":"Nex-N2-Pro","description":"Open agentic MoE model (397B total, 17B active) for coding, tool use, and research workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":262144},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}},"meituan-longcat/LongCat-2.0":{"id":"meituan-longcat/LongCat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1049000,"output":131072},"cost":{"input":0.75,"output":2.95,"cache_read":0.015}},"google/gemma-4-12B-it":{"id":"google/gemma-4-12B-it","name":"Gemma 4 12B IT","description":"Compact Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.4}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.4}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":1.6}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.24,"output":1.8}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.08}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.39,"output":2.34}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":131000},"cost":{"input":2,"output":6,"cache_read":0.25}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":3.2}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"deepseek-ai/DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.014}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.28,"cache_read":0.028}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.50162,"output":3.135,"cache_read":0.135}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42,"cache_read":0.135}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"deepseek-ai/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.41}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMaxAI/MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":197000,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.77,"output":3.4,"cache_read":0.14}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262000},"cost":{"input":2.7,"output":13.5,"cache_read":0.27}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.85916,"output":3.8,"cache_read":0.17993}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":131000},"cost":{"input":1.19,"output":3.74,"cache_read":0.6,"cache_write":0}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.302,"output":4.092,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":0.95,"output":2.55,"cache_read":0.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"zai-org/GLM-5V-Turbo":{"id":"zai-org/GLM-5V-Turbo","name":"zai-org/GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"openai/gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.04,"output":0.18}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"openai/gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.05,"output":0.45}}}},"minimax-coding-plan":{"id":"minimax-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax Token Plan (minimax.io)","doc":"https://platform.minimax.io/docs/token-plan/intro","models":{"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"alibaba-coding-plan":{"id":"alibaba-coding-plan","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-intl.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/coding-plan","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"lilac":{"id":"lilac","env":["LILAC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.getlilac.com/v1","name":"Lilac","doc":"https://docs.getlilac.com/inference/models","models":{"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":262100},"cost":{"input":0.11,"output":0.35}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.28,"output":1.1,"cache_read":0.05}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.2}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":524288},"cost":{"input":0.9,"output":3,"cache_read":0.27}}}},"moonshotai-cn":{"id":"moonshotai-cn","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.cn/v1","name":"Moonshot AI (China)","doc":"https://platform.moonshot.cn/docs/api/chat","models":{"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"bothub":{"id":"bothub","env":["BOTHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.bothub.ru/v1","name":"Bothub","doc":"https://bothub.ru/models","models":{"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.44}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1,"output":0.28}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.61,"output":4.84}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.06,"output":0.37}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.72,"output":5.41}}}},"xiaomi-token-plan-cn":{"id":"xiaomi-token-plan-cn","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-cn.xiaomimimo.com/v1","name":"Xiaomi Token Plan (China)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}}}},"privatemode-ai":{"id":"privatemode-ai","env":["PRIVATEMODE_API_KEY","PRIVATEMODE_ENDPOINT"],"npm":"@ai-sdk/openai-compatible","api":"http://localhost:8080/v1","name":"Privatemode AI","doc":"https://docs.privatemode.ai/api/overview","models":{"glm-flash-latest":{"id":"glm-flash-latest","name":"GLM Flash (latest)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":0.8897,"output":4.4718,"cache_read":0.0924}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":0.8897,"output":4.4718,"cache_read":0.0924}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper large-v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.01618,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"status":"deprecated","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"beta","cost":{"input":0.8897,"output":1.4675,"cache_read":0.0924}},"qwen3-embedding-4b":{"id":"qwen3-embedding-4b","name":"Qwen3-Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-06","last_updated":"2025-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2560},"cost":{"input":0.1502,"output":0}},"voxtral-mini-3b":{"id":"voxtral-mini-3b","name":"Voxtral Mini 3B","description":"Speech-to-text model for audio transcription, translation, and audio understanding","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07","last_updated":"2025-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.00462,"output":0}},"kimi-latest":{"id":"kimi-latest","name":"Kimi (latest)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"status":"deprecated","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.4969,"output":1.9644,"cache_read":0.0462}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"glm-latest":{"id":"glm-latest","name":"GLM (latest)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}}}},"llmgateway-providers":{"id":"llmgateway-providers","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"LLM Gateway","doc":"https://llmgateway.io/docs","models":{"consensusprotocol/gemma-4-31b-it":{"id":"consensusprotocol/gemma-4-31b-it","name":"Gemma 4 31B IT (Consensus Protocol)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"consensusprotocol/Qwen3.8-27B":{"id":"consensusprotocol/Qwen3.8-27B","name":"Qwen3.8 27B (Consensus Protocol)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.08,"output":0.35,"cache_read":0.05}},"consensusprotocol/glm-5.3-flash":{"id":"consensusprotocol/glm-5.3-flash","name":"GLM-5.3 Flash (Consensus Protocol)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.25,"cache_read":0.02}},"consensusprotocol/deepseek-v4.1-flash":{"id":"consensusprotocol/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Consensus Protocol)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.005}},"consensusprotocol/gpt-oss-20b":{"id":"consensusprotocol/gpt-oss-20b","name":"GPT OSS 20B (Consensus Protocol)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"consensusprotocol/deepseek-v4-flash":{"id":"consensusprotocol/deepseek-v4-flash","name":"DeepSeek V4 Flash (Consensus Protocol)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"deepinfra/gemma-4-31b-it":{"id":"deepinfra/gemma-4-31b-it","name":"Gemma 4 31B IT (DeepInfra)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"deepinfra/qwen3-vl-235b-a22b-instruct":{"id":"deepinfra/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (DeepInfra)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"deepinfra/mimo-v2.5":{"id":"deepinfra/mimo-v2.5","name":"MiMo V2.5 (DeepInfra)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"deepinfra/qwen3.8-27b":{"id":"deepinfra/qwen3.8-27b","name":"Qwen3.8 27B (DeepInfra)","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.2,"output":2.5,"cache_read":0.05}},"deepinfra/nemotron-3.5-lightning":{"id":"deepinfra/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning (DeepInfra)","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.08,"output":0.2}},"deepinfra/qwen3.8-2.4t-a95b":{"id":"deepinfra/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B (DeepInfra)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.2}},"deepinfra/muse-glimmer-30b":{"id":"deepinfra/muse-glimmer-30b","name":"Muse Glimmer 30B (DeepInfra)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"deepinfra/deepseek-v4.1-flash":{"id":"deepinfra/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (DeepInfra)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepinfra/ling-3.0-flash":{"id":"deepinfra/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (DeepInfra)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"deepinfra/qwen3-vl-30b-a3b-instruct":{"id":"deepinfra/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (DeepInfra)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":0.6}},"deepinfra/gemma-4-26b-a4b-it":{"id":"deepinfra/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (DeepInfra)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"deepinfra/nemotron-3-ultra-550b":{"id":"deepinfra/nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B (DeepInfra)","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/inkling-small":{"id":"deepinfra/inkling-small","name":"Inkling Small (DeepInfra)","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"deepinfra/mimo-v2.5-pro":{"id":"deepinfra/mimo-v2.5-pro","name":"MiMo V2.5 Pro (DeepInfra)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"deepinfra/qwen3.5-9b":{"id":"deepinfra/qwen3.5-9b","name":"Qwen3.5 9B (DeepInfra)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.15}},"deepinfra/inkling":{"id":"deepinfra/inkling","name":"Inkling (DeepInfra)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":262144},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"deepinfra/hy3":{"id":"deepinfra/hy3","name":"Hy3 (DeepInfra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"deepinfra/glm-5.1":{"id":"deepinfra/glm-5.1","name":"GLM-5.1 (DeepInfra)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":65536},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"deepinfra/deepseek-v4-pro":{"id":"deepinfra/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepInfra)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/step-3.7-flash":{"id":"deepinfra/step-3.7-flash","name":"Step 3.7 Flash (DeepInfra)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepinfra/deepseek-v3.2":{"id":"deepinfra/deepseek-v3.2","name":"DeepSeek V3.2 (DeepInfra)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":65536},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepinfra/glm-5.3":{"id":"deepinfra/glm-5.3","name":"GLM-5.3 (DeepInfra)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"deepinfra/deepseek-v4-flash":{"id":"deepinfra/deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepInfra)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.08,"output":0.18,"cache_read":0.016}},"cerebras/gemma-4-31b-it":{"id":"cerebras/gemma-4-31b-it","name":"Gemma 4 31B IT (Cerebras)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.99,"output":1.49}},"cerebras/qwen3-235b-a22b-instruct-2507":{"id":"cerebras/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Cerebras)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.6,"output":1.2}},"cerebras/llama-3.3-70b-instruct":{"id":"cerebras/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (Cerebras)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.85,"output":1.2}},"cerebras/glm-4.7":{"id":"cerebras/glm-4.7","name":"GLM-4.7 (Cerebras)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":2.25,"output":2.75}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.1,"output":0.5}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.15,"output":0.75}},"scx-ai-gp/glm-5.3-flash":{"id":"scx-ai-gp/glm-5.3-flash","name":"GLM-5.3 Flash (SCX.ai)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.088,"output":0.25,"cache_read":0.025}},"scx-ai-gp/glm-5.2-fast":{"id":"scx-ai-gp/glm-5.2-fast","name":"GLM-5.2 Turbo (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"scx-ai-gp/qwen3.8-max":{"id":"scx-ai-gp/qwen3.8-max","name":"Qwen3.8 Max (SCX.ai)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"scx-ai-gp/kimi-k3":{"id":"scx-ai-gp/kimi-k3","name":"Kimi K3 (SCX.ai)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3.5,"output":18,"cache_read":0.35}},"scx-ai-gp/deepseek-v4.1-flash":{"id":"scx-ai-gp/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (SCX.ai)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.01}},"scx-ai-gp/glm-5.2":{"id":"scx-ai-gp/glm-5.2","name":"GLM-5.2 (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.88,"output":2.55,"cache_read":0.16}},"scx-ai-gp/glm-5.3":{"id":"scx-ai-gp/glm-5.3","name":"GLM-5.3 (SCX.ai)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"scx-ai-gp/kimi-k2.7-code":{"id":"scx-ai-gp/kimi-k2.7-code","name":"Kimi K2.7 Code (SCX.ai)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5 (Z AI)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash (Z AI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6 (Z AI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5-x":{"id":"zai/glm-4.5-x","name":"GLM-4.5 X (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5 (Z AI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-4.5-airx":{"id":"zai/glm-4.5-airx","name":"GLM-4.5 AirX (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM-4.5V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-4-32b-0414-128k":{"id":"zai/glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k) (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"zai/glm-4.6v-flashx":{"id":"zai/glm-4.6v-flashx","name":"GLM-4.6V FlashX (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7 (Z AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2 (Z AI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1 (Z AI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX (Z AI)","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air (Z AI)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3 (Z AI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"aws-bedrock/claude-haiku-4-5":{"id":"aws-bedrock/claude-haiku-4-5","name":"Claude Haiku 4.5 (AWS Bedrock)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/llama-4-maverick-17b-instruct":{"id":"aws-bedrock/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (AWS Bedrock)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.24,"output":0.97}},"aws-bedrock/claude-sonnet-4-5":{"id":"aws-bedrock/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/claude-fable-5-1":{"id":"aws-bedrock/claude-fable-5-1","name":"Claude Fable 5.1 (AWS Bedrock)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"aws-bedrock/llama-4-scout-17b-instruct":{"id":"aws-bedrock/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (AWS Bedrock)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.17,"output":0.66}},"aws-bedrock/claude-opus-4-1-20250805":{"id":"aws-bedrock/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"aws-bedrock/claude-opus-4-5-20251101":{"id":"aws-bedrock/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-opus-5":{"id":"aws-bedrock/claude-opus-5","name":"Claude Opus 5 (AWS Bedrock)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-fable-5":{"id":"aws-bedrock/claude-fable-5","name":"Claude Fable 5 (AWS Bedrock)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"aws-bedrock/claude-opus-4-8":{"id":"aws-bedrock/claude-opus-4-8","name":"Claude Opus 4.8 (AWS Bedrock)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-sonnet-4-5-20250929":{"id":"aws-bedrock/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/claude-sonnet-5":{"id":"aws-bedrock/claude-sonnet-5","name":"Claude Sonnet 5 (AWS Bedrock)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"aws-bedrock/claude-opus-4-6":{"id":"aws-bedrock/claude-opus-4-6","name":"Claude Opus 4.6 (AWS Bedrock)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/grok-4-3":{"id":"aws-bedrock/grok-4-3","name":"Grok 4.3 (AWS Bedrock)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"aws-bedrock/claude-haiku-4-5-20251001":{"id":"aws-bedrock/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (AWS Bedrock)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/grok-4-6":{"id":"aws-bedrock/grok-4-6","name":"Grok 4.6 (AWS Bedrock)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"aws-bedrock/claude-sonnet-4-6":{"id":"aws-bedrock/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (AWS Bedrock)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/claude-opus-4-7":{"id":"aws-bedrock/claude-opus-4-7","name":"Claude Opus 4.7 (AWS Bedrock)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/llama-3.1-70b-instruct":{"id":"aws-bedrock/llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct (AWS Bedrock)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.72,"output":0.72}},"google-ai-studio/gemini-3.6-flash":{"id":"google-ai-studio/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3.5-flash-lite":{"id":"google-ai-studio/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-ai-studio/gemini-3.1-pro-preview":{"id":"google-ai-studio/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google AI Studio)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-ai-studio/gemini-3.5-flash":{"id":"google-ai-studio/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-ai-studio/gemini-2.5-pro":{"id":"google-ai-studio/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google AI Studio)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-ai-studio/gemini-2.5-flash":{"id":"google-ai-studio/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google AI Studio)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google-ai-studio/gemini-3.7-flash":{"id":"google-ai-studio/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google AI Studio)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3-flash-preview":{"id":"google-ai-studio/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google AI Studio)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-ai-studio/gemini-pro-latest":{"id":"google-ai-studio/gemini-pro-latest","name":"Gemini Pro Latest (Google AI Studio)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google-ai-studio/gemini-3.8-flash":{"id":"google-ai-studio/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google AI Studio)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-2.5-flash-lite":{"id":"google-ai-studio/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google AI Studio)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-ai-studio/gemini-3.1-flash-lite":{"id":"google-ai-studio/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google AI Studio)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Anthropic)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5-5":{"id":"anthropic/claude-opus-5-5","name":"Claude Opus 5.5 (Anthropic)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1 (Anthropic)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Anthropic)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5 (Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5 (Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (Anthropic)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Anthropic)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (DeepSeek)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"runpod/kimi-k3":{"id":"runpod/kimi-k3","name":"Kimi K3 (Runpod)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"runware/gemma-4-31b-it":{"id":"runware/gemma-4-31b-it","name":"Gemma 4 31B IT (Runware)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.102,"output":0.297,"cache_read":0.012}},"runware/glm-5.3-flash":{"id":"runware/glm-5.3-flash","name":"GLM-5.3 Flash (Runware)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"runware/kimi-k3":{"id":"runware/kimi-k3","name":"Kimi K3 (Runware)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"runware/deepseek-v4.1-flash":{"id":"runware/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Runware)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.01}},"runware/kimi-k2.6":{"id":"runware/kimi-k2.6","name":"Kimi K2.6 (Runware)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"runware/glm-5.2":{"id":"runware/glm-5.2","name":"GLM-5.2 (Runware)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"runware/deepseek-v4-pro":{"id":"runware/deepseek-v4-pro","name":"DeepSeek V4 Pro (Runware)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.961,"output":1.922,"cache_read":0.079}},"runware/gpt-oss-120b":{"id":"runware/gpt-oss-120b","name":"GPT OSS 120B (Runware)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"runware/glm-5.3":{"id":"runware/glm-5.3","name":"GLM-5.3 (Runware)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"runware/deepseek-v4-flash":{"id":"runware/deepseek-v4-flash","name":"DeepSeek V4 Flash (Runware)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"tencent/glm-5v-turbo":{"id":"tencent/glm-5v-turbo","name":"GLM-5V Turbo (Tencent Cloud)","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"tencent/kimi-k3":{"id":"tencent/kimi-k3","name":"Kimi K3 (Tencent Cloud)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"tencent/glm-5":{"id":"tencent/glm-5","name":"GLM-5 (Tencent Cloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"tencent/kimi-k2.7-code-highspeed":{"id":"tencent/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed (Tencent Cloud)","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"tencent/kimi-k2.6":{"id":"tencent/kimi-k2.6","name":"Kimi K2.6 (Tencent Cloud)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.858,"output":3.566,"cache_read":0.145}},"tencent/minimax-m2.7":{"id":"tencent/minimax-m2.7","name":"MiniMax M2.7 (Tencent Cloud)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"tencent/mimo-v2.5-pro":{"id":"tencent/mimo-v2.5-pro","name":"MiMo V2.5 Pro (Tencent Cloud)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"tencent/minimax-m3":{"id":"tencent/minimax-m3","name":"MiniMax M3 (Tencent Cloud)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"tencent/glm-5.2":{"id":"tencent/glm-5.2","name":"GLM-5.2 (Tencent Cloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3 (Tencent Cloud)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}},"tencent/hy-mt2-plus":{"id":"tencent/hy-mt2-plus","name":"Hy-MT2 Plus (Tencent Cloud)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/glm-5.1":{"id":"tencent/glm-5.1","name":"GLM-5.1 (Tencent Cloud)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"tencent/glm-5-turbo":{"id":"tencent/glm-5-turbo","name":"GLM-5 Turbo (Tencent Cloud)","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"tencent/deepseek-v4-pro":{"id":"tencent/deepseek-v4-pro","name":"DeepSeek V4 Pro (Tencent Cloud)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.00363}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 Preview (Tencent Cloud)","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/kimi-k2.7-code":{"id":"tencent/kimi-k2.7-code","name":"Kimi K2.7 Code (Tencent Cloud)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"tencent/deepseek-v4-flash":{"id":"tencent/deepseek-v4-flash","name":"DeepSeek V4 Flash (Tencent Cloud)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"meta-contributor/muse-spark-1.2-contributor":{"id":"meta-contributor/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta-contributor/muse-spark-1.3-contributor":{"id":"meta-contributor/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"GLM-4.6V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"novita/gemma-4-31b-it":{"id":"novita/gemma-4-31b-it","name":"Gemma 4 31B IT (NovitaAI)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"novita/qwen3-vl-235b-a22b-instruct":{"id":"novita/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (NovitaAI)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"novita/qwen3.7-max":{"id":"novita/qwen3.7-max","name":"Qwen3.7 Max (NovitaAI)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"novita/mimo-v2.5":{"id":"novita/mimo-v2.5","name":"MiMo V2.5 (NovitaAI)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.168,"output":0.336,"cache_read":0.0034,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"novita/qwen3.8-27b":{"id":"novita/qwen3.8-27b","name":"Qwen3.8 27B (NovitaAI)","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"novita/qwen3-vl-235b-a22b-thinking":{"id":"novita/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking (NovitaAI)","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"novita/qwen3.8-2.4t-a95b":{"id":"novita/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B (NovitaAI)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"novita/glm-5.3-flash":{"id":"novita/glm-5.3-flash","name":"GLM-5.3 Flash (NovitaAI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6 (NovitaAI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"novita/qwen3-235b-a22b-instruct-2507":{"id":"novita/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (NovitaAI)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"novita/qwen3.8-max":{"id":"novita/qwen3.8-max","name":"Qwen3.8 Max (NovitaAI)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"novita/kimi-k3":{"id":"novita/kimi-k3","name":"Kimi K3 (NovitaAI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"novita/llama-4-maverick-17b-instruct":{"id":"novita/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (NovitaAI)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5 (NovitaAI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/deepseek-v4.1-flash":{"id":"novita/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (NovitaAI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"novita/minimax-m2.5":{"id":"novita/minimax-m2.5","name":"MiniMax M2.5 (NovitaAI)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi K2.6 (NovitaAI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"novita/qwen3-max":{"id":"novita/qwen3-max","name":"Qwen3 Max (NovitaAI)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38}},"novita/glm-4.5v":{"id":"novita/glm-4.5v","name":"GLM-4.5V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"novita/llama-4-scout-17b-instruct":{"id":"novita/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (NovitaAI)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"novita/qwen3-235b-a22b-fp8":{"id":"novita/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8 (NovitaAI)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"novita/qwen3-coder-480b-a35b-instruct":{"id":"novita/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (NovitaAI)","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"novita/qwen3-coder-30b-a3b-instruct":{"id":"novita/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct (NovitaAI)","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"novita/ling-3.0-flash":{"id":"novita/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (NovitaAI)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"novita/qwen3-vl-30b-a3b-instruct":{"id":"novita/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (NovitaAI)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"novita/gemma-4-26b-a4b-it":{"id":"novita/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (NovitaAI)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"novita/llama-3-70b-instruct":{"id":"novita/llama-3-70b-instruct","name":"Llama 3 70B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"novita/qwen3-235b-a22b-thinking-2507":{"id":"novita/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507 (NovitaAI)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"novita/llama-3.3-70b-instruct":{"id":"novita/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (NovitaAI)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"novita/qwen3.6-35b-a3b":{"id":"novita/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (NovitaAI)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.248,"output":1.485}},"novita/qwen3-next-80b-a3b-instruct":{"id":"novita/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (NovitaAI)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"novita/minimax-m2.7":{"id":"novita/minimax-m2.7","name":"MiniMax M2.7 (NovitaAI)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"novita/mimo-v2.5-pro":{"id":"novita/mimo-v2.5-pro","name":"MiMo V2.5 Pro (NovitaAI)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"GLM-4.7 (NovitaAI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"novita/qwen3.8-flash":{"id":"novita/qwen3.8-flash","name":"Qwen3.8 Flash (NovitaAI)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"novita/qwen35-397b-a17b":{"id":"novita/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"novita/glm-5.2":{"id":"novita/glm-5.2","name":"GLM-5.2 (NovitaAI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/ernie-4.5-vl-424b-a47b":{"id":"novita/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"novita/hy3":{"id":"novita/hy3","name":"Hy3 (NovitaAI)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"novita/kimi-k2":{"id":"novita/kimi-k2","name":"Kimi K2 (NovitaAI)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"novita/glm-5.1":{"id":"novita/glm-5.1","name":"GLM-5.1 (NovitaAI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"novita/step-3.7-flash":{"id":"novita/step-3.7-flash","name":"Step 3.7 Flash (NovitaAI)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"novita/llama-3.2-3b-instruct":{"id":"novita/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek V3.2 (NovitaAI)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"novita/glm-5.3":{"id":"novita/glm-5.3","name":"GLM-5.3 (NovitaAI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/kimi-k2.7-code":{"id":"novita/kimi-k2.7-code","name":"Kimi K2.7 Code (NovitaAI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"MiniMax M2.1 (NovitaAI)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/deepseek-v4-flash":{"id":"novita/deepseek-v4-flash","name":"DeepSeek V4 Flash (NovitaAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"embercloud/glm-4.5":{"id":"embercloud/glm-4.5","name":"GLM-4.5 (EmberCloud)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"embercloud/qwen3-coder-next":{"id":"embercloud/qwen3-coder-next","name":"Qwen3 Coder Next (EmberCloud)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"embercloud/glm-5":{"id":"embercloud/glm-5","name":"GLM-5 (EmberCloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.72,"output":2.3,"cache_read":0.144}},"embercloud/kimi-k2.5":{"id":"embercloud/kimi-k2.5","name":"Kimi K2.5 (EmberCloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"embercloud/glm-4.7-flash":{"id":"embercloud/glm-4.7-flash","name":"GLM-4.7 Flash (EmberCloud)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"embercloud/glm-4.7":{"id":"embercloud/glm-4.7","name":"GLM-4.7 (EmberCloud)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.38,"output":1.98,"cache_read":0.19}},"embercloud/glm-5.2":{"id":"embercloud/glm-5.2","name":"GLM-5.2 (EmberCloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"embercloud/glm-5.1":{"id":"embercloud/glm-5.1","name":"GLM-5.1 (EmberCloud)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.931,"output":2.93,"cache_read":0.173}},"embercloud/glm-4.5-air":{"id":"embercloud/glm-4.5-air","name":"GLM-4.5 Air (EmberCloud)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro (Perplexity)","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar (Perplexity)","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro (Perplexity)","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"azure-ai-foundry/grok-4-1-fast-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-1-fast-non-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-3":{"id":"azure-ai-foundry/grok-4-3","name":"Grok 4.3 (Azure AI Foundry)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":8192},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3 (Meta)","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1 (Meta)","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2 (Meta)","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"azure/gpt-5.4":{"id":"azure/gpt-5.4","name":"GPT-5.4 (Azure)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"azure/gpt-5.4-pro":{"id":"azure/gpt-5.4-pro","name":"GPT-5.4 Pro (Azure)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"azure/gpt-3.5-turbo":{"id":"azure/gpt-3.5-turbo","name":"GPT-3.5 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"azure/gpt-5.4-nano":{"id":"azure/gpt-5.4-nano","name":"GPT-5.4 Nano (Azure)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":1.25,"output":10}},"azure/gpt-4o":{"id":"azure/gpt-4o","name":"GPT-4o (Azure)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"azure/gpt-5-mini":{"id":"azure/gpt-5-mini","name":"GPT-5 Mini (Azure)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-5.2-pro":{"id":"azure/gpt-5.2-pro","name":"GPT-5.2 Pro (Azure)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"azure/o4-mini":{"id":"azure/o4-mini","name":"o4 Mini (Azure)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"azure/o3-mini":{"id":"azure/o3-mini","name":"o3 Mini (Azure)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"azure/gpt-4":{"id":"azure/gpt-4","name":"GPT-4 (Azure)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"azure/gpt-5.3-codex":{"id":"azure/gpt-5.3-codex","name":"GPT-5.3 Codex (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-4.1-nano":{"id":"azure/gpt-4.1-nano","name":"GPT-4.1 Nano (Azure)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"azure/gpt-5-nano":{"id":"azure/gpt-5-nano","name":"GPT-5 Nano (Azure)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"azure/o1":{"id":"azure/o1","name":"o1 (Azure)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"azure/gpt-6-astra":{"id":"azure/gpt-6-astra","name":"GPT-6 Astra (Azure)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure/gpt-5.1":{"id":"azure/gpt-5.1","name":"GPT-5.1 (Azure)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.4-mini":{"id":"azure/gpt-5.4-mini","name":"GPT-5.4 Mini (Azure)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"azure/gpt-5.6-luna":{"id":"azure/gpt-5.6-luna","name":"GPT-5.6 Luna (Azure)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"azure/gpt-5.2":{"id":"azure/gpt-5.2","name":"GPT-5.2 (Azure)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5.5":{"id":"azure/gpt-5.5","name":"GPT-5.5 (Azure)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"azure/gpt-4.1":{"id":"azure/gpt-4.1","name":"GPT-4.1 (Azure)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-4.1-mini":{"id":"azure/gpt-4.1-mini","name":"GPT-4.1 Mini (Azure)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"azure/gpt-6-luna":{"id":"azure/gpt-6-luna","name":"GPT-6 Luna (Azure)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"azure/gpt-5.6-terra":{"id":"azure/gpt-5.6-terra","name":"GPT-5.6 Terra (Azure)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"azure/gpt-4-turbo":{"id":"azure/gpt-4-turbo","name":"GPT-4 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"azure/gpt-oss-120b":{"id":"azure/gpt-oss-120b","name":"GPT OSS 120B (Azure)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/o3":{"id":"azure/o3","name":"o3 (Azure)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-5":{"id":"azure/gpt-5","name":"GPT-5 (Azure)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.6-sol":{"id":"azure/gpt-5.6-sol","name":"GPT-5.6 Sol (Azure)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"azure/gpt-6-sol":{"id":"azure/gpt-6-sol","name":"GPT-6 Sol (Azure)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"alibaba/qwen-flash":{"id":"alibaba/qwen-flash","name":"Qwen Flash (Alibaba Cloud)","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max (Alibaba Cloud)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen-coder-plus":{"id":"alibaba/qwen-coder-plus","name":"Qwen Coder Plus (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"alibaba/qwen3-vl-plus":{"id":"alibaba/qwen3-vl-plus","name":"Qwen3 VL Plus (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"cache_read":0.04,"cache_write":0.25}},"alibaba/qwen-max":{"id":"alibaba/qwen-max","name":"Qwen Max (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max (Alibaba Cloud)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/kimi-k3":{"id":"alibaba/kimi-k3","name":"Kimi K3 (Alibaba Cloud)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"alibaba/glm-5":{"id":"alibaba/glm-5","name":"GLM-5 (Alibaba Cloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58}},"alibaba/deepseek-v4.1-flash":{"id":"alibaba/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Alibaba Cloud)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"alibaba/qwen-plus":{"id":"alibaba/qwen-plus","name":"Qwen Plus (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen3.7 Flash (Alibaba Cloud)","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max (Alibaba Cloud)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32800},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"alibaba/qwen3-vl-flash":{"id":"alibaba/qwen3-vl-flash","name":"Qwen3 VL Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"alibaba/kimi-k2.5":{"id":"alibaba/kimi-k2.5","name":"Kimi K2.5 (Alibaba Cloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.574,"output":3.011}},"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (Alibaba Cloud)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.375,"output":2.25}},"alibaba/qwen3.6-flash":{"id":"alibaba/qwen3.6-flash","name":"Qwen3.6 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"alibaba/qwen3-coder-flash":{"id":"alibaba/qwen3-coder-flash","name":"Qwen3 Coder Flash (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus (Alibaba Cloud)","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":66000},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen3.8 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/qwen35-397b-a17b":{"id":"alibaba/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3.6-max-preview":{"id":"alibaba/qwen3.6-max-preview","name":"Qwen3.6 Max Preview (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13}},"alibaba/glm-5.2":{"id":"alibaba/glm-5.2","name":"GLM-5.2 (Alibaba Cloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28}},"alibaba/qwen-omni-turbo":{"id":"alibaba/qwen-omni-turbo","name":"Qwen Omni Turbo (Alibaba Cloud)","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.2,"output":0.8}},"alibaba/deepseek-v4-pro":{"id":"alibaba/deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen3.6 Plus (Alibaba Cloud)","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"alibaba/qwen-plus-latest":{"id":"alibaba/qwen-plus-latest","name":"Qwen Plus Latest (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-09","last_updated":"2024-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"alibaba/glm-5.3":{"id":"alibaba/glm-5.3","name":"GLM-5.3 (Alibaba Cloud)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus (Alibaba Cloud)","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"alibaba/deepseek-v4-flash":{"id":"alibaba/deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"baidu/glm-5":{"id":"baidu/glm-5","name":"GLM-5 (Baidu)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"baidu/deepseek-v4.1-flash":{"id":"baidu/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Baidu)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"baidu/kimi-k2.6":{"id":"baidu/kimi-k2.6","name":"Kimi K2.6 (Baidu)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"baidu/glm-5.2":{"id":"baidu/glm-5.2","name":"GLM-5.2 (Baidu)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/glm-5.1":{"id":"baidu/glm-5.1","name":"GLM-5.1 (Baidu)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-pro":{"id":"baidu/deepseek-v4-pro","name":"DeepSeek V4 Pro (Baidu)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.042}},"baidu/glm-5.3":{"id":"baidu/glm-5.3","name":"GLM-5.3 (Baidu)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-flash":{"id":"baidu/deepseek-v4-flash","name":"DeepSeek V4 Flash (Baidu)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"gonka24/glm-5.3-flash":{"id":"gonka24/glm-5.3-flash","name":"GLM-5.3 Flash (Gonka24)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":16384},"cost":{"input":0.15,"output":0.3,"cache_read":0.035}},"gonka24/minimax-m2.7":{"id":"gonka24/minimax-m2.7","name":"MiniMax M2.7 (Gonka24)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.08,"output":0.32,"cache_read":0.017}},"gonka24/deepseek-v4-flash":{"id":"gonka24/deepseek-v4-flash","name":"DeepSeek V4 Flash (Gonka24)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":390000,"output":16384},"cost":{"input":0.065,"output":0.116,"cache_read":0.012}},"bytedance/seed-1-6-flash-250715":{"id":"bytedance/seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"bytedance/seed-1-6-250915":{"id":"bytedance/seed-1-6-250915","name":"Seed 1.6 (250915) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/seed-1-8-251228":{"id":"bytedance/seed-1-8-251228","name":"Seed 1.8 (251228) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/glm-4.7":{"id":"bytedance/glm-4.7","name":"GLM-4.7 (ByteDance)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"bytedance/seed-1-6-250615":{"id":"bytedance/seed-1-6-250615","name":"Seed 1.6 (250615) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/glm-5.2":{"id":"bytedance/glm-5.2","name":"GLM-5.2 (ByteDance)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"bytedance/deepseek-v4-pro":{"id":"bytedance/deepseek-v4-pro","name":"DeepSeek V4 Pro (ByteDance)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"bytedance/deepseek-v3.2":{"id":"bytedance/deepseek-v3.2","name":"DeepSeek V3.2 (ByteDance)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.28,"output":0.42,"cache_read":0.056}},"bytedance/gpt-oss-120b":{"id":"bytedance/gpt-oss-120b","name":"GPT OSS 120B (ByteDance)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.1,"output":0.5,"cache_read":0.02}},"bytedance/deepseek-v4-flash":{"id":"bytedance/deepseek-v4-flash","name":"DeepSeek V4 Flash (ByteDance)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"xai/grok-4-5":{"id":"xai/grok-4-5","name":"Grok 4.5 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-build-0-1":{"id":"xai/grok-build-0-1","name":"Grok Build 0.1 (xAI)","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok 4 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-20-beta-0309-non-reasoning":{"id":"xai/grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 Beta Non-Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-3":{"id":"xai/grok-4-3","name":"Grok 4.3 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-20-beta-0309-reasoning":{"id":"xai/grok-4-20-beta-0309-reasoning","name":"Grok 4.20 Beta Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-6":{"id":"xai/grok-4-6","name":"Grok 4.6 (xAI)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4-7":{"id":"xai/grok-4-7","name":"Grok 4.7 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"ranoai/deepseek-v4-flash":{"id":"ranoai/deepseek-v4-flash","name":"DeepSeek V4 Flash (RanoAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"canopywave/kimi-k3":{"id":"canopywave/kimi-k3","name":"Kimi K3 (CanopyWave)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"canopywave/kimi-k2.6":{"id":"canopywave/kimi-k2.6","name":"Kimi K2.6 (CanopyWave)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"canopywave/glm-5.2":{"id":"canopywave/glm-5.2","name":"GLM-5.2 (CanopyWave)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"canopywave/deepseek-v4-pro":{"id":"canopywave/deepseek-v4-pro","name":"DeepSeek V4 Pro (CanopyWave)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.74,"output":3.48,"cache_read":0.01}},"canopywave/deepseek-v4-flash":{"id":"canopywave/deepseek-v4-flash","name":"DeepSeek V4 Flash (CanopyWave)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"fireworks/kimi-k3":{"id":"fireworks/kimi-k3","name":"Kimi K3 (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":3,"output":15,"cache_read":0.3}},"fireworks/deepseek-v4.1-flash":{"id":"fireworks/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Fireworks AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks/kimi-k3-fast":{"id":"fireworks/kimi-k3-fast","name":"Kimi K3 Fast (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"fireworks/deepseek-v4-pro":{"id":"fireworks/deepseek-v4-pro","name":"DeepSeek V4 Pro (Fireworks AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"fireworks/deepseek-v4-flash":{"id":"fireworks/deepseek-v4-flash","name":"DeepSeek V4 Flash (Fireworks AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max (Sakana AI)","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2.0":{"id":"sakana/fugu-ultra-v2.0","name":"Fugu Ultra v2.0 (Sakana AI)","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra (Sakana AI)","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"scx-ai/gemma-4-31b-it":{"id":"scx-ai/gemma-4-31b-it","name":"Gemma 4 31B IT (SCX.ai (Turbo))","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.91}},"scx-ai/llama-4-maverick-17b-instruct":{"id":"scx-ai/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (SCX.ai (Turbo))","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.53,"output":1.62}},"scx-ai/qwen3-32b":{"id":"scx-ai/qwen3-32b","name":"Qwen3 32B (SCX.ai (Turbo))","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.36,"output":0.87}},"scx-ai/minimax-m2.7":{"id":"scx-ai/minimax-m2.7","name":"MiniMax M2.7 (SCX.ai (Turbo))","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}},"scx-ai/gpt-oss-120b":{"id":"scx-ai/gpt-oss-120b","name":"GPT OSS 120B (SCX.ai (Turbo))","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.17,"output":0.55}},"together-ai/qwen3.8-2.4t-a95b":{"id":"together-ai/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B (Together AI)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1010000,"output":909000},"cost":{"input":2,"output":6,"cache_read":0.25}},"together-ai/glm-5.3-flash":{"id":"together-ai/glm-5.3-flash","name":"GLM-5.3 Flash (Together AI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943717},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"together-ai/muse-glimmer-30b":{"id":"together-ai/muse-glimmer-30b","name":"Muse Glimmer 30B (Together AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"together-ai/kimi-k3":{"id":"together-ai/kimi-k3","name":"Kimi K3 (Together AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":3,"output":15,"cache_read":0.3}},"together-ai/deepseek-v4.1-flash":{"id":"together-ai/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Together AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"together-ai/inkling":{"id":"together-ai/inkling","name":"Inkling (Together AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":471859},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"together-ai/glm-4.7":{"id":"together-ai/glm-4.7","name":"GLM-4.7 (Together AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.45,"output":2}},"together-ai/minimax-m3":{"id":"together-ai/minimax-m3","name":"MiniMax M3 (Together AI)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"together-ai/deepseek-v4-pro":{"id":"together-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro (Together AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":163840},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together-ai/gpt-oss-120b":{"id":"together-ai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"together-ai/glm-5.3":{"id":"together-ai/glm-5.3","name":"GLM-5.3 (Together AI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943717},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"together-ai/deepseek-v4-flash":{"id":"together-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash (Together AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo V2.6 Pro (Xiaomi)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5 (Xiaomi)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro (Xiaomi)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo V2.6 Flash (Xiaomi)","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"quartz/gemini-3.1-pro-preview":{"id":"quartz/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Quartz)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5 (MiniMax)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"minimax/minimax-text-01":{"id":"minimax/minimax-text-01","name":"MiniMax Text 01 (MiniMax)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7 (MiniMax)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3 (MiniMax)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed (MiniMax)","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1 (MiniMax)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.27,"output":1.1}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2 (MiniMax)","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"inference.net/llama-3.2-11b-instruct":{"id":"inference.net/llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct (Inference.net)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.33}},"azure-anthropic/claude-opus-5":{"id":"azure-anthropic/claude-opus-5","name":"Claude Opus 5 (Azure Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-fable-5":{"id":"azure-anthropic/claude-fable-5","name":"Claude Fable 5 (Azure Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure-anthropic/claude-opus-4-8":{"id":"azure-anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Azure Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-sonnet-5":{"id":"azure-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Azure Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"azure-anthropic/claude-opus-4-6":{"id":"azure-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Azure Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-opus-4-7":{"id":"azure-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Azure Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"atria/atria-dawn-preview":{"id":"atria/atria-dawn-preview","name":"Atria Dawn Preview (Atria)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large Latest (Mistral AI)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"mistral/ministral-8b-2512":{"id":"mistral/ministral-8b-2512","name":"Ministral 8B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.15,"output":0.15}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2 (Mistral AI)","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3 (Mistral AI)","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/ministral-14b-2512":{"id":"mistral/ministral-14b-2512","name":"Ministral 14B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.2}},"mistral/mistral-small-2506":{"id":"mistral/mistral-small-2506","name":"Mistral Small 3.2 (Mistral AI)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"mistral/ministral-3b-2512":{"id":"mistral/ministral-3b-2512","name":"Ministral 3B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1}},"mistral/codestral-2508":{"id":"mistral/codestral-2508","name":"Codestral (Mistral AI)","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"vertex-anthropic/claude-haiku-4-5":{"id":"vertex-anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Vertex AI (Anthropic))","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"vertex-anthropic/claude-sonnet-4-5":{"id":"vertex-anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Vertex AI (Anthropic))","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-opus-4-5-20251101":{"id":"vertex-anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Vertex AI (Anthropic))","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-sonnet-5":{"id":"vertex-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Vertex AI (Anthropic))","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"vertex-anthropic/claude-opus-4-6":{"id":"vertex-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Vertex AI (Anthropic))","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-sonnet-4-6":{"id":"vertex-anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Vertex AI (Anthropic))","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-opus-4-7":{"id":"vertex-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Vertex AI (Anthropic))","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-openai/qwen3-235b-a22b-instruct-2507":{"id":"vertex-openai/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Vertex AI (OpenAI-compatible))","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.22,"output":0.88}},"vertex-openai/qwen3-next-80b-a3b-thinking":{"id":"vertex-openai/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking (Vertex AI (OpenAI-compatible))","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/glm-5":{"id":"vertex-openai/glm-5","name":"GLM-5 (Vertex AI (OpenAI-compatible))","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"vertex-openai/qwen3-coder-480b-a35b-instruct":{"id":"vertex-openai/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (Vertex AI (OpenAI-compatible))","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"vertex-openai/kimi-k2-thinking":{"id":"vertex-openai/kimi-k2-thinking","name":"Kimi K2 Thinking (Vertex AI (OpenAI-compatible))","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"vertex-openai/qwen3-next-80b-a3b-instruct":{"id":"vertex-openai/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (Vertex AI (OpenAI-compatible))","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/glm-4.7":{"id":"vertex-openai/glm-4.7","name":"GLM-4.7 (Vertex AI (OpenAI-compatible))","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.6,"output":2.2}},"vertex-openai/grok-4-20-non-reasoning":{"id":"vertex-openai/grok-4-20-non-reasoning","name":"Grok 4.20 Non-Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"vertex-openai/grok-4-6":{"id":"vertex-openai/grok-4-6","name":"Grok 4.6 (Vertex AI (OpenAI-compatible))","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"vertex-openai/grok-4-20-reasoning":{"id":"vertex-openai/grok-4-20-reasoning","name":"Grok 4.20 Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"vertex-openai/deepseek-v3.2":{"id":"vertex-openai/deepseek-v3.2","name":"DeepSeek V3.2 (Vertex AI (OpenAI-compatible))","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"google-vertex/gemini-3.6-flash":{"id":"google-vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3.5-flash-lite":{"id":"google-vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-vertex/gemini-3.1-pro-preview":{"id":"google-vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-vertex/gemini-3.5-flash":{"id":"google-vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-vertex/gemini-2.5-pro":{"id":"google-vertex/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google Vertex AI)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-vertex/gemini-2.5-flash":{"id":"google-vertex/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google Vertex AI)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google-vertex/gemini-3.7-flash":{"id":"google-vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3-flash-preview":{"id":"google-vertex/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-vertex/gemini-3.8-flash":{"id":"google-vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-2.5-flash-lite":{"id":"google-vertex/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google Vertex AI)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-vertex/gemini-3.1-flash-lite":{"id":"google-vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"aws-mantle/gpt-6-astra":{"id":"aws-mantle/gpt-6-astra","name":"GPT-6 Astra (AWS Mantle)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"aws-mantle/gpt-5.6-luna":{"id":"aws-mantle/gpt-5.6-luna","name":"GPT-5.6 Luna (AWS Mantle)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"aws-mantle/gpt-5.6-terra":{"id":"aws-mantle/gpt-5.6-terra","name":"GPT-5.6 Terra (AWS Mantle)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75}},"aws-mantle/gpt-5.6-sol":{"id":"aws-mantle/gpt-5.6-sol","name":"GPT-5.6 Sol (AWS Mantle)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":921600,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4 (OpenAI)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro (OpenAI)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":2.5,"output":10}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro (OpenAI)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano (OpenAI)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o (OpenAI)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini (OpenAI)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro (OpenAI)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini (OpenAI)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini (OpenAI)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4 (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex (OpenAI)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano (OpenAI)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano (OpenAI)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"o1 (OpenAI)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro (OpenAI)","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1 (OpenAI)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini (OpenAI)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini (OpenAI)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna (OpenAI)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2 (OpenAI)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5 (OpenAI)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1 (OpenAI)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini (OpenAI)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna (OpenAI)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":1.25,"output":5}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra (OpenAI)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/o3":{"id":"openai/o3","name":"o3 (OpenAI)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5 (OpenAI)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol (OpenAI)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol (OpenAI)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3 (Moonshot AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed (Moonshot AI)","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6 (Moonshot AI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5 (Moonshot AI)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code (Moonshot AI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"above":{"id":"above","env":["ABOVE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.above.dev/v1","name":"above.dev","doc":"https://above.dev/docs","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo V2.6 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.5077,"output":1.0154,"cache_read":0.0042}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.165,"output":0.55,"cache_read":0.0319}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.31,"output":7.26,"cache_read":0.231}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen 3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.2,"output":6.6,"cache_read":0.275}},"mimo-v2.6-pro-ultraspeed":{"id":"mimo-v2.6-pro-ultraspeed","name":"MiMo V2.6 Pro UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":5.0769,"output":10.1538,"cache_read":0.0423}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo V2.6 Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1692,"output":0.3385,"cache_read":0.0034}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.54,"output":4.84,"cache_read":0.154}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.726,"output":2.178,"reasoning":2.178,"cache_read":0.0242}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.165,"output":0.66,"reasoning":0.66,"cache_read":0.0033}}}},"kilo":{"id":"kilo","env":["KILO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kilo.ai/api/gateway","name":"Kilo Gateway","doc":"https://kilo.ai","models":{"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stealth/claude-opus-4.6":{"id":"stealth/claude-opus-4.6","name":"Stealth: Claude Opus 4.6 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/claude-opus-4.7":{"id":"stealth/claude-opus-4.7","name":"Stealth: Claude Opus 4.7 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/claude-opus-4.8":{"id":"stealth/claude-opus-4.8","name":"Stealth: Claude Opus 4.8 (20% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Claude Opus 4.8 is offered at 20% lower cost than standard Claude Opus 4.8 pricing and is not served by Anthropic or Kilo Code.","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/space-bunny-alpha":{"id":"stealth/space-bunny-alpha","name":"Space Bunny Alpha","description":"Space Bunny Alpha is an anonymous large model with blazing-fast inference, strong coding capabilities and native multimodal input support. It delivers adjustable reasoning effort, and a 1M-token context window. Space...","family":"alpha","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":524288},"cost":{"input":0,"output":0}},"stealth/claude-sonnet-4.6":{"id":"stealth/claude-sonnet-4.6","name":"Stealth: Claude Sonnet 4.6 (20% off)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.4,"output":12,"reasoning":0,"cache_read":0.24,"cache_write":3}},"stealth/qwen3.6-plus":{"id":"stealth/qwen3.6-plus","name":"Stealth: Qwen3.6 Plus (50% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Qwen3.6 Plus is offered at 50% lower cost than standard Qwen3.6 Plus pricing and is not served by Alibaba or Kilo Code. Note: a surcharge applies to long-context workloads exceeding 256K input tokens.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":0,"cache_read":0.025,"cache_write":0.3125}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed: Seed 2.1 Turbo","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"ByteDance Seed: Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"MoonshotAI: Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.35,"output":11.5,"cache_read":0.3}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Poolside: Laguna S 2.1 (free)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Poolside: Laguna XS 2.1 (free)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"PrismML: Ternary Bonsai 2 27B","description":"Bonsai 2 27B is a 27B-parameter reasoning model from PrismML derived from Qwen3.8-27B. It supports coding, mathematics, tool calling, and image understanding with a 262K-token context window. Ternary compression shrinks...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.5}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude Opus 5.5 is Anthropic's flagship model for demanding reasoning, coding, and long-horizon agentic work, succeeding Claude Opus 5. It is particularly strong at multi-step changes in large codebases, code...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex AGI: Nex-N2.5-Mini (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex AGI: Nex-N2.5-Pro (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"cohere/command-a-plus":{"id":"cohere/command-a-plus","name":"Cohere: Command A+","description":"Command A+ is Cohere's flagship model for enterprise agentic workflows. It accepts text and image inputs with a 192K context window, supports native tool calling with strict tool schemas, structured...","family":"command-a","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":64000},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"cohere/command-a":{"id":"cohere/command-a","name":"Cohere: Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"Cohere: North Mini Code (free)","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the cost-efficient tier of the V4.1 family. DeepSeek reports that it exceeds V4 Pro on performance, speed, and task...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B (retires Sep 28)","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus (retires Sep 28)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.32,"output":0.89}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp (retires Sep 28)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":147456},"cost":{"input":0.25,"output":1}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek: R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Tencent: Hy-MT2-30B-A3B","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Tencent: Hy-MT2-1.8B","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Tencent: Hy-MT2-7B","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Meta: Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Meta: Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1875,"output":0.6525}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.02,"output":0.04}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Google: Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Google: Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron: Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5.3-prime":{"id":"z-ai/glm-5.3-prime","name":"Z.ai: GLM 5.3 Prime","description":"GLM-5.3-Prime is the high-speed variant of Z.ai's GLM-5.3, inheriting its full capabilities while delivering 1.5–2× the output throughput through inference acceleration. It supports text input and output with a 1M-token...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.8,"output":8.8,"cache_read":0.56}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"Z.ai: GLM 5.3 FlashX","description":"GLM-5.3-FlashX is the high-speed variant of Z.ai's GLM-5.3-Flash, a native multimodal model delivering inference speeds of up to 200 tokens/s. Built on the same hybrid sparse and linear attention architecture...","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.09}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.0605,"output":0.4}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.2:free":{"id":"z-ai/glm-5.2:free","name":"Z.ai: GLM 5.2 (free)","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0,"output":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":943717},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Inference.net: Schematron V2 Small","description":"Schematron V2 Small is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes extraction quality for complex schemas and long pages. Extraction instructions must be supplied through a JSON schema...","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Inference.net: Schematron V2 Turbo","description":"Schematron V2 Turbo is a 3B-parameter HTML-to-JSON extraction model from Inference.net. It prioritizes throughput for high-volume extraction workloads. Extraction instructions must be supplied through a JSON schema in response_format rather...","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Thinking Machines: Inkling Small (free)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":471859},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"status":"beta","cost":{"input":0,"output":0}},"openrouter/free":{"id":"openrouter/free","name":"OpenRouter Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":0,"output":0}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["audio","image","pdf","text","video"],"output":["image","text"]},"open_weights":false,"limit":{"context":2000000,"output":32768},"cost":{"input":0,"output":0}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Perplexity: Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Perplexity: Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Perplexity: Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Perplexity: Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.3,"output":1.1,"cache_read":0.04}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Meta: Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Meta: Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Nous: Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nousresearch","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"Z.ai: GLM Flash Latest","description":"This model always redirects to the latest model in the GLM Flash family.","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":0.045,"output":0.14,"cache_read":0.01}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"Z.ai: GLM Latest","description":"This model always redirects to the latest GLM model from Z.ai.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.5614,"output":1.7644,"cache_read":0.10426}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"~openai/gpt-terra-latest":{"id":"~openai/gpt-terra-latest","name":"OpenAI: GPT Terra Latest","description":"This model always redirects to the latest model in the OpenAI GPT Terra family.","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"~openai/gpt-luna-latest":{"id":"~openai/gpt-luna-latest","name":"OpenAI: GPT Luna Latest","description":"This model always redirects to the latest model in the OpenAI GPT Luna family.","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"~openai/gpt-sol-latest":{"id":"~openai/gpt-sol-latest","name":"OpenAI: GPT Sol Latest","description":"This model always redirects to the latest model in the OpenAI GPT Sol family.","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"~openai/gpt-astra-latest":{"id":"~openai/gpt-astra-latest","name":"OpenAI: GPT Astra Latest ($$$$)","description":"This model always redirects to the latest model in the OpenAI GPT Astra family.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"OpenAI: GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B (retires Oct 8)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"Grok 4.7 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"SpaceXAI: Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Anthropic: Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Anthropic: Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Anthropic: Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Anthropic: Claude Fable Latest ($$$$)","description":"This model always redirects to the latest model in the Claude Fable family.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Upstage: Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Upstage: Solar Pro 4","description":"Solar Pro 4 is a large language model from Upstage. It is suited for agentic workflows, office productivity, document-intensive work, and coding.","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"upstage/solar-mini4":{"id":"upstage/solar-mini4","name":"Upstage: Solar Mini 4","description":"Solar Mini 4 is Upstage's compact, cost-efficient language model, a 35B-parameter mixture-of-experts with 3B active parameters and a 524K context window. It is built for agentic use cases where response...","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"~deepseek/deepseek-flash-latest":{"id":"~deepseek/deepseek-flash-latest","name":"DeepSeek: DeepSeek Flash Latest","description":"This model always redirects to the latest model in the DeepSeek Flash family.","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.04,"output":1,"cache_read":0.01}},"~deepseek/deepseek-pro-latest":{"id":"~deepseek/deepseek-pro-latest","name":"DeepSeek: DeepSeek Pro Latest","description":"This model always redirects to the latest model in the DeepSeek Pro family.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3894,"output":1.1682,"cache_read":0.01239}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek: DeepSeek V4 Flash Latest","description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.03,"output":0.32,"cache_read":0.016}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"xAI: Grok Latest","description":"This model always redirects to the latest Grok model from xAI.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LiquidAI: LFM2.5-2.6B (free)","description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.16}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.375,"output":1.875,"reasoning":1.875,"cache_read":0.0375,"cache_write":0.020833}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.042,"output":0.22}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.15,"output":1.25,"reasoning":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":4.5,"reasoning":4.5,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.041667}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows.","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Writer: Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"IBM: Granite 4.2 8B","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"fireworks/ember-1":{"id":"fireworks/ember-1","name":"Fireworks: Ember-1","description":"Ember-1 is a specialized reasoning model from Fireworks Research, built on [Kimi K3](https://openrouter.ai/moonshotai/kimi-k3). It is designed to make every token go further: it produces shorter reasoning traces, using roughly 40%...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-24","last_updated":"2026-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral: Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral: Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Mistral: Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Sakana: Fugu Max","description":"Fugu Max is the cost-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Sakana: Fugu Ultra v2","description":"Fugu Ultra v2 is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"inclusionAI: Ling 3.0 Flash Fin (free)","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"inclusionAI: Ling 3.0 Flash Sante (free)","description":"Ling 3.0 Flash Sante is a health and medicine-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"inclusionAI: Ling 3.0 Flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"inclusionAI: Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"inclusionAI: Ling 3.0 Flash Fin","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.6,"output":3}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"MoonshotAI: Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.6562,"output":3.3,"cache_read":0.18}},"kilo-auto/small":{"id":"kilo-auto/small","name":"Auto Small","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"reasoning":0,"cache_read":0.005}},"kilo-auto/efficient":{"id":"kilo-auto/efficient","name":"Auto Efficient","description":"Routes each request to the cheapest model that gets the job done, based on continuously benchmarked accuracy and cost.","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"kilo-auto/free":{"id":"kilo-auto/free","name":"Auto Free","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0,"cache_write":0}},"kilo-auto/frontier":{"id":"kilo-auto/frontier","name":"Auto Frontier","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"reasoning":0,"cache_read":0.5,"cache_write":6.25}},"kilo-auto/balanced":{"id":"kilo-auto/balanced","name":"Auto Balanced","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"NVIDIA: Nemotron 3.5 Content Safety (free)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.065,"output":0.18}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"NVIDIA: Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.08,"output":0.45}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"NVIDIA: Nemotron 3 Ultra (free)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":182520},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"NVIDIA: Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"NVIDIA: Nemotron 3.5 Lightning (free)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"MiMo-V2.6-Pro is the flagship foundation model developed by Xiaomi. Built at a scale of over 1T parameters, it is designed to push the ceiling of capability for the most demanding...","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.6-pro-ultraspeed":{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","name":"MiMo-V2.6-Pro-UltraSpeed","description":"MiMo-V2.6-Pro-UltraSpeed is the fast speed edition of Xiaomi's flagship foundation model, MiMo-V2.6-Pro. Built from the same 1T MiMo-V2.6-Pro checkpoint, it matches the original model in quality while delivering roughly 10x...","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.004,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo-V2.6-Flash is an open-source foundation model developed by Xiaomi. Built on a Mixture-of-Experts architecture with 309B total parameters and 15B activated per token, it employs a hybrid attention mechanism for...","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3686},"cost":{"input":0.08,"output":0.11}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax: MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"output":900172},"cost":{"input":0.2,"output":1.1}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax: MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.4,"output":2.2}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":176947},"cost":{"input":0.3,"output":1.2}},"mancer/weaver":{"id":"mancer/weaver","name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"stepfun/step-3.7-flash:free":{"id":"stepfun/step-3.7-flash:free","name":"StepFun: Step 3.7 Flash (free)","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters per token. The model supports a 256K context window and exposes selectable reasoning levels (high/medium/low), letting callers trade off speed, cost, and depth of reasoning. Designed for coding, agentic workflows, structured outputs, and long-context productivity tasks.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots Studio: Dots3-Note Preview (free)","description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Inception: Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.2,"output":0.75,"cache_read":0.02}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Inception: Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon: Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon: Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon: Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Amazon: Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Amazon: Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace: Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace: Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5":{"id":"aion-labs/aion-3.5","name":"AionLabs: Aion 3.5","description":"Aion 3.5 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5-mini":{"id":"aion-labs/aion-3.5-mini","name":"AionLabs: Aion 3.5 Mini","description":"Aion 3.5 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It is the smaller, lower-cost sibling of Aion 3.5 and uses...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.1495,"output":0.598}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.26,"output":1.04}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.475,"output":4.425,"cache_read":0.295,"cache_write":1.84375}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.425,"output":2.55,"cache_read":0.085,"cache_write":0.53125}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.975,"output":4.875}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.2925,"output":1.4625}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen3.8 Omni Flash is an omni-modal reasoning model from Alibaba, the first Qwen model built around agentic capabilities with native audio-video understanding. It is suited for audio-video analysis and summarization,...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.13,"output":0.52}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.0975,"output":0.78}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3.8-max-prime":{"id":"qwen/qwen3.8-max-prime","name":"Qwen 3.8 Max Prime","description":"Qwen3.8 Max Prime is a higher-throughput variant of Qwen3.8 Max from Alibaba's Qwen team, served as a separate SKU at a higher price point. It accepts text, image, and video...","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":4,"output":12,"cache_read":0.5}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1625,"output":1.3}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.13,"output":0.52}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.39,"output":2.34}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen: Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.13,"output":0.52}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262140},"cost":{"input":0.45,"output":2.7}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen: Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.2275,"output":0.91}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4}},"qwen/qwen3.8-27b:free":{"id":"qwen/qwen3.8-27b:free","name":"Qwen: Qwen3.8 27B (free)","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph: Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph: Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"OpenAI: GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"OpenAI: GPT-6 Astra Pro ($$$$)","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"OpenAI: GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"OpenAI: GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.018,"output":0.09}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI: o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-sol-pro":{"id":"openai/gpt-6-sol-pro","name":"OpenAI: GPT-6 Sol Pro","description":"GPT-6 Sol Pro is the same underlying model as [GPT-6 Sol](https://openrouter.ai/openai/gpt-6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"OpenAI: GPT-5 Image ($$$$)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"GPT-6 Luna is the fast, cost-efficient model in OpenAI's GPT-6 series, positioned below GPT-6 Sol. It is suited for high-volume and latency-sensitive workloads such as chat, classification, and lightweight agentic...","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI: o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-6-luna-pro":{"id":"openai/gpt-6-luna-pro","name":"OpenAI: GPT-6 Luna Pro","description":"GPT-6 Luna Pro is the same underlying model as [GPT-6 Luna](https://openrouter.ai/openai/gpt-6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"GPT-6 Sol is the cost-efficient high-end model in OpenAI's GPT-6 series, positioned below the flagship GPT-6 Astra and above the fast GPT-6 Luna tier. It is suited for demanding professional...","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Microsoft: Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}}}},"venice":{"id":"venice","env":["VENICE_API_KEY"],"npm":"venice-ai-sdk-provider","name":"Venice AI","doc":"https://docs.venice.ai","models":{"qwen-3-8-2-4t-a95b":{"id":"qwen-3-8-2-4t-a95b","name":"Qwen 3.8 2.4T","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125}},"openai-gpt-56-luna":{"id":"openai-gpt-56-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32000},"cost":{"input":2.27,"output":6.8,"cache_read":0.34,"tiers":[{"input":4.53,"output":13.6,"cache_read":0.68,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":0.68}}},"aion-labs-aion-3-5":{"id":"aion-labs-aion-3-5","name":"Aion 3.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":3.75,"output":7.5,"cache_read":0.9375}},"kimi-k3-fast-api":{"id":"kimi-k3-fast-api","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.325,"output":3.25}},"hermes-3-llama-3.1-405b":{"id":"hermes-3-llama-3.1-405b","name":"Hermes 3 Llama 3.1 405b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"hermes","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-09-25","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":3}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-13","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"google-gemma-3-27b-it":{"id":"google-gemma-3-27b-it","name":"Google Gemma 3 27B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-04","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.12,"output":0.2}},"qwen-3-7-max":{"id":"qwen-3-7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.7,"output":8.05,"cache_read":0.27,"cache_write":3.35}},"qwen-3-8-max":{"id":"qwen-3-8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-22","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125,"cache_write":3.125}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.175,"output":0.35,"cache_read":0.035}},"venice-uncensored-role-play":{"id":"venice-uncensored-role-play","name":"Venice Role Play Uncensored","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":2}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen 3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.75}},"openai-gpt-6-luna":{"id":"openai-gpt-6-luna","name":"GPT-6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.125,"output":0.625,"cache_read":0.0125,"cache_write":0.15625,"tiers":[{"input":0.25,"output":0.9375,"cache_read":0.025,"cache_write":0.3125,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.25,"output":0.9375,"cache_read":0.025,"cache_write":0.3125}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3.75,"output":18.75,"cache_read":0.375}},"openai-gpt-54-mini":{"id":"openai-gpt-54-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.9375,"output":5.625,"cache_read":0.09375}},"qwen-3-8-flash":{"id":"qwen-3-8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.49,"cache_read":0.014}},"olafangensan-glm-4.7-flash-heretic":{"id":"olafangensan-glm-4.7-flash-heretic","name":"GLM 4.7 Flash Heretic","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":24000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"deepseek-v4-flash-0731-fast":{"id":"deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731 Fast","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-09","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.35,"output":0.7,"cache_read":0.0875}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-06","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":32768},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"zai-org-glm-5-2":{"id":"zai-org-glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"claude-opus-4-8-fast":{"id":"claude-opus-4-8-fast","name":"Claude Opus 4.8 Fast","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"z-ai-glm-5-turbo":{"id":"z-ai-glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai-org-glm-4.6":{"id":"zai-org-glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2024-04-01","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"openai-gpt-54-pro":{"id":"openai-gpt-54-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225,"tiers":[{"input":75,"output":337.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":75,"output":337.5}}},"aion-labs-aion-3-0":{"id":"aion-labs-aion-3-0","name":"Aion 3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":3.75,"output":7.5,"cache_read":0.9375}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":64000},"cost":{"input":3.75,"output":18.75,"cache_read":0.375,"cache_write":4.69}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-18","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.8,"output":24,"cache_read":0.24,"cache_write":6}},"kimi-k2-5":{"id":"kimi-k2-5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-04","release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.56,"output":3.5,"cache_read":0.22}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-08-29","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":0.3,"cache_write":15}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"grok-4-20":{"id":"grok-4-20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"openai-gpt-56-sol-pro":{"id":"openai-gpt-56-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-28","last_updated":"2026-07-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.625,"output":3.125,"cache_read":0.125}},"z-ai-glm-5v-turbo":{"id":"z-ai-glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":1.5,"output":5,"cache_read":0.3}},"openai-gpt-55-pro":{"id":"openai-gpt-55-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225}},"qwen-3-7-plus":{"id":"qwen-3-7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875}}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen 3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.45,"output":3.5}},"zai-org-glm-4.7-flash":{"id":"zai-org-glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-05","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"minimax-m27":{"id":"minimax-m27","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.375,"output":1.5,"cache_read":0.06875}},"mercury-2-5":{"id":"mercury-2-5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-08","last_updated":"2026-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04999999999999999,"output":0.18749999999999994,"cache_read":0.004999999999999999}},"qwen3-next-80b":{"id":"qwen3-next-80b","name":"Qwen 3 Next 80b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.35,"output":1.9}},"minimax-m25":{"id":"minimax-m25","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.27,"output":0.95,"cache_read":0.03}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen 3.6 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.1,"output":1}},"grok-4-20-multi-agent":{"id":"grok-4-20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"zai-org-glm-4.7":{"id":"zai-org-glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.55,"output":2.65,"cache_read":0.11}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen 3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.15}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3 VL 235B","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-10","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":4.95,"cache_read":0.165}},"openai-gpt-56-terra":{"id":"openai-gpt-56-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"llama-3.3-70b":{"id":"llama-3.3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"release_date":"2025-04-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.7,"output":2.8}},"openai-gpt-4o-mini-2024-07-18":{"id":"openai-gpt-4o-mini-2024-07-18","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1875,"output":0.75,"cache_read":0.09375}},"mistral-small-3-2-24b-instruct":{"id":"mistral-small-3-2-24b-instruct","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.25,"output":5.0625,"cache_read":0.2125}},"z-ai-glm-5-3":{"id":"z-ai-glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.75,"output":5.5,"cache_read":0.325}},"google-gemma-4-26b-a4b-it":{"id":"google-gemma-4-26b-a4b-it","name":"Google Gemma 4 26B A4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.13,"output":0.4,"cache_read":0.05}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"openai-gpt-6-astra-pro":{"id":"openai-gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-05","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":12.5,"output":62.5,"cache_read":1.25,"cache_write":15.625,"tiers":[{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25}}},"openai-gpt-54":{"id":"openai-gpt-54","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":3.13,"output":18.8,"cache_read":0.313}},"gemma-4-uncensored":{"id":"gemma-4-uncensored","name":"Gemma 4 Uncensored","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-13","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1625,"output":0.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"minimax-m3-preview":{"id":"minimax-m3-preview","name":"MiniMax M3 Preview","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-12","last_updated":"2026-06-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-07-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.375,"output":1.5,"cache_read":0.0075}},"gemini-3-8-flash":{"id":"gemini-3-8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus Uncensored","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-06","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.625,"output":3.75,"cache_read":0.0625,"cache_write":0.78,"tiers":[{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78}}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"openai-gpt-6-sol":{"id":"openai-gpt-6-sol","name":"GPT-6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":12.5,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25}}},"qwen-3-8-27b":{"id":"qwen-3-8-27b","name":"Qwen 3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-17","last_updated":"2026-08-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.2}},"openai-gpt-53-codex":{"id":"openai-gpt-53-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"aion-labs-aion-3-0-mini":{"id":"aion-labs-aion-3-0-mini","name":"Aion 3.0 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.875,"output":1.75,"cache_read":0.225}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-18","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"aion-labs-aion-3-5-mini":{"id":"aion-labs-aion-3-5-mini","name":"Aion 3.5 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.875,"output":1.75,"cache_read":0.225}},"openai-gpt-56-terra-pro":{"id":"openai-gpt-56-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-19","last_updated":"2026-06-11","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.7,"output":3.75,"cache_read":0.07}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.3125,"output":0.9375,"cache_read":0.03125}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-10","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":200000},"cost":{"input":2.27,"output":6.8,"cache_read":0.57,"tiers":[{"input":4.53,"output":13.6,"cache_read":1.13,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":1.13}}},"zai-org-glm-5-1":{"id":"zai-org-glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":80000},"cost":{"input":1.54,"output":4.84,"cache_read":0.286}},"xiaomi-mimo-v2-5":{"id":"xiaomi-mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-06-11","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"openai-gpt-4o-2024-11-20":{"id":"openai-gpt-4o-2024-11-20","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":3.125,"output":12.5}},"openai-gpt-52":{"id":"openai-gpt-52","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-13","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":272000,"output":65536},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.5,"output":15,"cache_read":0.5,"cache_write":0.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":3.301,"cache_read":0.33}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.6,"output":18,"cache_read":0.36,"cache_write":4.5}},"claude-opus-5-fast":{"id":"claude-opus-5-fast","name":"Claude Opus 5 Fast","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"openai-gpt-55":{"id":"openai-gpt-55","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":6.25,"output":37.5,"cache_read":0.625,"tiers":[{"input":12.5,"output":56.25,"cache_read":1.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":12.5,"output":56.25,"cache_read":1.25}}},"openai-gpt-56-sol":{"id":"openai-gpt-56-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"qwen3-coder-480b-a35b-instruct-turbo":{"id":"qwen3-coder-480b-a35b-instruct-turbo","name":"Qwen 3 Coder 480B Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"z-ai-glm-5-3-flash":{"id":"z-ai-glm-5-3-flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai-org-glm-5":{"id":"zai-org-glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"grok-4-7":{"id":"grok-4-7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-16","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":200000},"cost":{"input":2.27,"output":6.8,"cache_read":0.57,"tiers":[{"input":4.53,"output":13.6,"cache_read":1.13,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":1.13}}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.33,"output":0.48,"cache_read":0.16}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.375,"output":3.125,"cache_read":0.0375}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.55,"output":9.45,"cache_read":0.155,"cache_write":0.086}},"google-gemma-4-31b-it":{"id":"google-gemma-4-31b-it","name":"Google Gemma 4 31B Instruct","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-03","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.12,"output":0.36,"cache_read":0.09}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"venice-uncensored-1-2":{"id":"venice-uncensored-1-2","name":"Venice Uncensored 1.2","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen 3.5 397B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.75,"output":4.5}},"llama-3.2-3b":{"id":"llama-3.2-3b","name":"Llama 3.2 3B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-10-03","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.6}},"openai-gpt-56-luna-pro":{"id":"openai-gpt-56-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"nvidia-nemotron-3-ultra-550b-a55b":{"id":"nvidia-nemotron-3-ultra-550b-a55b","name":"NVIDIA Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.625,"output":3.125,"cache_read":0.1875}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash 0423","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.138,"output":0.275,"cache_read":0.028}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"NVIDIA Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.3}}}},"alibaba-token-plan-cn":{"id":"alibaba-token-plan-cn","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan (China)","doc":"https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview","models":{"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"ai21":{"id":"ai21","env":["AI21_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ai21.com/studio/v1","name":"AI21 Labs","doc":"https://docs.ai21.com/docs/jamba-foundation-models","models":{"jamba-large":{"id":"jamba-large","name":"Jamba Large","description":"AI21's hybrid SSM-Transformer long-context model for enterprise agents and grounded generation","family":"jamba","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-22","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":2,"output":8}},"jamba-mini":{"id":"jamba-mini","name":"Jamba Mini","description":"AI21's efficient, lightweight hybrid SSM-Transformer model for a wide range of tasks","family":"jamba","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-22","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.2,"output":0.4}}}},"inference":{"id":"inference","env":["INFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.net/v1","name":"Inference","doc":"https://inference.net/models","models":{"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.025,"output":0.025}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.055,"output":0.055}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.01,"output":0.01}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.02,"output":0.02}},"google/gemma-3":{"id":"google/gemma-3","name":"Google Gemma 3","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.15,"output":0.3}},"osmosis/osmosis-structure-0.6b":{"id":"osmosis/osmosis-structure-0.6b","name":"Osmosis Structure 0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"osmosis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":2048},"cost":{"input":0.1,"output":0.5}},"mistral/mistral-nemo-12b-instruct":{"id":"mistral/mistral-nemo-12b-instruct","name":"Mistral Nemo 12B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.038,"output":0.1}},"qwen/qwen3-embedding-4b":{"id":"qwen/qwen3-embedding-4b","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"qwen/qwen-2.5-7b-vision-instruct":{"id":"qwen/qwen-2.5-7b-vision-instruct","name":"Qwen 2.5 7B Vision Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.2,"output":0.2}}}},"iflowcn":{"id":"iflowcn","env":["IFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apis.iflow.cn/v1","name":"iFlow","doc":"https://platform.iflow.cn/en/docs","models":{"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL-Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3-235B-A22B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3-Max-Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-instruct":{"id":"qwen3-235b-a22b-instruct","name":"Qwen3-235B-A22B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi-K2-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"kimi-k2":{"id":"kimi-k2","name":"Kimi-K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-235b":{"id":"qwen3-235b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}}}},"minimax-cn-coding-plan":{"id":"minimax-cn-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.cn/anthropic/v1","name":"MiniMax Token Plan (minimax.cn)","doc":"https://platform.minimaxi.com/docs/token-plan/intro","models":{"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"thinkingmachines":{"id":"thinkingmachines","env":["TINKER_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://tinker.thinkingmachines.dev/services/tinker-prod/anthropic/api/v1","name":"Thinking Machines","doc":"https://tinker-docs.thinkingmachines.ai/tinker/compatible-apis/anthropic/","models":{"thinkingmachines/Inkling:peft:262144":{"id":"thinkingmachines/Inkling:peft:262144","name":"Inkling (256K)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}}}},"stepfun-step-plan":{"id":"stepfun-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/step_plan/v1","name":"StepFun Step Plan (China)","doc":"https://platform.stepfun.com/docs/zh/step-plan/integrations/reasoning-api","models":{"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536}},"step-router-v1":{"id":"step-router-v1","name":"Step Router v1","description":"StepFun routing model that dispatches requests to the appropriate Step model.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}}}},"melious":{"id":"melious","env":["MELIOUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.melious.ai/v1","name":"Melious","doc":"https://melious.ai/docs/reference/models","models":{"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11592,"output":0.46368,"cache_read":0.023184}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.11592,"output":0.2898,"cache_read":0.023184}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3.1878,"output":15.939,"cache_read":0.788256}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":1.10124,"output":3.36168,"cache_read":0.266616}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.23184,"output":1.1592,"cache_read":0.011592}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.81144,"output":4.0572,"cache_read":0.266616}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.5796,"output":2.95596,"cache_read":0.139104}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.1592,"output":3.4776,"cache_read":0.11592}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1592,"output":4.6368,"cache_read":0.2898}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":1.50696,"output":4.6368,"cache_read":0.370944}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.85472,"output":3.70944,"cache_read":0.46368}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":32768},"cost":{"input":0.69552,"output":2.78208,"cache_read":0.185472}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":64000},"cost":{"input":0.34776,"output":0.5796,"cache_read":0.092736}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1592,"output":3.4776,"cache_read":0.23184}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.81144,"output":3.4776,"cache_read":0.220248}}}},"berget":{"id":"berget","env":["BERGET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.berget.ai/v1","name":"Berget.AI","doc":"https://api.berget.ai","models":{"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["audio","image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.275,"output":0.55}},"Qwen/Qwen3.8-27B-FP8":{"id":"Qwen/Qwen3.8-27B-FP8","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.46,"output":3.48}},"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct 2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.33,"output":0.33}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":32768},"cost":{"input":3,"output":15}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":32768},"cost":{"input":1.54,"output":4.84}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":16384},"cost":{"input":0.29,"output":0.58}}}},"snowflake-cortex":{"id":"snowflake-cortex","env":["SNOWFLAKE_ACCOUNT","SNOWFLAKE_CORTEX_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1","name":"Snowflake Cortex","doc":"https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api","models":{"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192},"status":"beta"},"openai-gpt-5":{"id":"openai-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192},"status":"beta"},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"mistral-large2":{"id":"mistral-large2","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192},"status":"beta"},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384}},"openai-gpt-5.1":{"id":"openai-gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192}},"snowflake-llama3.3-70b":{"id":"snowflake-llama3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}}}},"sarvam":{"id":"sarvam","env":["SARVAM_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sarvam.ai/v1","name":"Sarvam AI","doc":"https://docs.sarvam.ai/api-reference-docs/getting-started/models","models":{"sarvam-30b":{"id":"sarvam-30b","name":"Sarvam-30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536}},"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam-105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}}}},"nova":{"id":"nova","env":["NOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nova.amazon.com/v1","name":"Nova","doc":"https://nova.amazon.com/dev/documentation","models":{"nova-2-pro-v1":{"id":"nova-2-pro-v1","name":"Nova 2 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-01-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}},"nova-2-lite-v1":{"id":"nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}}}},"abacus":{"id":"abacus","env":["ABACUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://routellm.abacus.ai/v1","name":"Abacus","doc":"https://abacus.ai/help/api","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"claude-3-7-sonnet-20250219":{"id":"claude-3-7-sonnet-20250219","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"qwen-2.5-coder-32b":{"id":"qwen-2.5-coder-32b","name":"Qwen 2.5 Coder 32B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.79,"output":0.79}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1.2,"output":6}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"kimi-k2-turbo-preview":{"id":"kimi-k2-turbo-preview","name":"Kimi K2 Turbo Preview","description":"Fast Kimi model for responsive chat, coding help, and agent loops","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":0.15,"output":8}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"Grok 4 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.59,"output":0.79}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":40}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.2,"output":1.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.3-codex-xhigh":{"id":"gpt-5.3-codex-xhigh","name":"GPT-5.3 Codex XHigh","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.5}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"route-llm":{"id":"route-llm","name":"RouteLLM","description":"RouteLLM routes prompts to an appropriate Abacus-backed text-generation model","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":3,"output":15}},"grok-4-0709":{"id":"grok-4-0709","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":3,"output":15}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.55,"output":1.66}},"meta-llama/Meta-Llama-3.3-70B-Instruct":{"id":"meta-llama/Meta-Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.59,"output":0.79}},"meta-llama/Meta-Llama-3.1-8B-Instruct":{"id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.05}},"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"id":"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo","name":"Llama 3.1 405B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":3.5,"output":3.5}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.14,"output":0.59}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"Qwen/QwQ-32B":{"id":"Qwen/QwQ-32B","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.4,"output":0.4}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":0.38}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.09,"output":0.29}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.29,"output":1.2}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.32,"output":3.2}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.74,"output":3.48,"cache_read":0.15}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-15","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":0.4}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":1}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":96000},"cost":{"input":0.6,"output":2.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.08,"output":0.44}}}},"novita-ai":{"id":"novita-ai","env":["NOVITA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.novita.ai/openai","name":"NovitaAI","doc":"https://novita.ai/docs/guides/introduction","models":{"kwaipilot/kat-coder-pro":{"id":"kwaipilot/kat-coder-pro","name":"Kat Coder Pro","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-05","last_updated":"2026-01-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"deepseek/deepseek-prover-v2-671b":{"id":"deepseek/deepseek-prover-v2-671b","name":"Deepseek Prover V2 671B","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":160000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-r1-distill-qwen-14b":{"id":"deepseek/deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.15,"output":0.15}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12,"cache_read":0.135}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill LLama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3-turbo":{"id":"deepseek/deepseek-v3-turbo","name":"DeepSeek V3 (Turbo)\t","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.4,"output":1.3}},"deepseek/deepseek-r1-0528-qwen3-8b":{"id":"deepseek/deepseek-r1-0528-qwen3-8b","name":"DeepSeek R1 0528 Qwen3 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.06,"output":0.09}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"Deepseek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-ocr-2":{"id":"deepseek/deepseek-ocr-2","name":"deepseek/deepseek-ocr-2","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-ocr":{"id":"deepseek/deepseek-ocr","name":"DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-r1-turbo":{"id":"deepseek/deepseek-r1-turbo","name":"DeepSeek R1 (Turbo)\t","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"Deepseek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-r1-distill-qwen-32b":{"id":"deepseek/deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32000},"cost":{"input":0.3,"output":0.3}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"Deepseek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"meta-llama/llama-3-70b-instruct":{"id":"meta-llama/llama-3-70b-instruct","name":"Llama3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-07","last_updated":"2024-12-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.05}},"meta-llama/llama-4-scout-17b-16e-instruct":{"id":"meta-llama/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"meta-llama/llama-3-8b-instruct":{"id":"meta-llama/llama-3-8b-instruct","name":"Llama 3 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.04,"output":0.04}},"nousresearch/hermes-2-pro-llama-3-8b":{"id":"nousresearch/hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-06-27","last_updated":"2024-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3,"cache_read":0.3}},"xiaomimimo/mimo-v2-pro":{"id":"xiaomimimo/mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.4,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomimimo/mimo-v2.5-pro":{"id":"xiaomimimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":0.522,"output":1.044,"cache_read":0.0043,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.522,"output":1.044,"cache_read":0.0043}}},"baidu/ernie-4.5-vl-28b-a3b":{"id":"baidu/ernie-4.5-vl-28b-a3b","name":"ERNIE 4.5 VL 28B A3B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2026-06-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":8000},"cost":{"input":0.14,"output":0.56}},"baidu/ernie-4.5-vl-28b-a3b-thinking":{"id":"baidu/ernie-4.5-vl-28b-a3b-thinking","name":"ERNIE-4.5-VL-28B-A3B-Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.39,"output":0.39}},"baidu/ernie-4.5-21B-a3b-thinking":{"id":"baidu/ernie-4.5-21B-a3b-thinking","name":"ERNIE-4.5-21B-A3B-Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-21B-a3b":{"id":"baidu/ernie-4.5-21B-a3b","name":"ERNIE 4.5 21B A3B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":8000},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":98304,"output":16384},"cost":{"input":0.119,"output":0.2}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.4}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.05,"output":0.1}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-30","last_updated":"2024-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60288,"output":16000},"cost":{"input":0.04,"output":0.17}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-08","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-1t":{"id":"inclusionai/ling-2.6-1t","name":"Ling-2.6-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-23","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-flash":{"id":"inclusionai/ling-2.6-flash","name":"Ling-2.6-flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-07","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"zai-org/glm-4.6v":{"id":"zai-org/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.6":{"id":"zai-org/glm-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"zai-org/glm-5":{"id":"zai-org/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"zai-org/autoglm-phone-9b-multilingual":{"id":"zai-org/autoglm-phone-9b-multilingual","name":"AutoGLM-Phone-9B-Multilingual","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.035,"output":0.138}},"zai-org/glm-4.5-air":{"id":"zai-org/glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"Mythomax L2 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3200},"cost":{"input":0.09,"output":0.09}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"paddlepaddle/paddleocr-vl":{"id":"paddlepaddle/paddleocr-vl","name":"PaddleOCR-VL","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.02}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-omni-30b-a3b-instruct":{"id":"qwen/qwen3-omni-30b-a3b-instruct","name":"Qwen3 Omni 30B A3B Instruct","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","video","audio","image"],"output":["text","audio"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen3-omni-30b-a3b-thinking":{"id":"qwen/qwen3-omni-30b-a3b-thinking","name":"Qwen3 Omni 30B A3B Thinking","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","audio","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.11,"output":8.45}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30b A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"qwen/qwen3-vl-30b-a3b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"qwen/qwen2.5-7b-instruct":{"id":"qwen/qwen2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.07,"output":0.07}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"qwen/qwen3-vl-30b-a3b-thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":1}},"qwen/qwen-mt-plus":{"id":"qwen/qwen-mt-plus","name":"Qwen MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-03","last_updated":"2025-09-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.25,"output":0.75}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}},"qwen/qwen3-8b-fp8":{"id":"qwen/qwen3-8b-fp8","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.035,"output":0.138}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"qwen/qwen3-vl-8b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.08,"output":0.5}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.8,"output":0.8}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.38,"output":0.4}},"qwen/qwen3-4b-fp8":{"id":"qwen/qwen3-4b-fp8","name":"Qwen3 4B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.03,"output":0.03}},"baichuan/baichuan-m2-32b":{"id":"baichuan/baichuan-m2-32b","name":"baichuan-m2-32b","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"baichuan","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.07,"output":0.07}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"OpenAI: GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.15}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.25}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"Wizardlm 2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-24","last_updated":"2024-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"sao10K/l3-70b-euryale-v2.1":{"id":"sao10K/l3-70b-euryale-v2.1","name":"L3 70B Euryale V2.1\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-18","last_updated":"2024-06-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}},"sao10K/L3-8B-stheno-v3.2":{"id":"sao10K/L3-8B-stheno-v3.2","name":"L3 8B Stheno V3.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-29","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":32000},"cost":{"input":0.05,"output":0.05}},"sao10K/l3-8b-lunaris":{"id":"sao10K/l3-8b-lunaris","name":"Sao10k L3 8B Lunaris\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.05,"output":0.05}},"sao10K/l31-70b-euryale-v2.2":{"id":"sao10K/l31-70b-euryale-v2.2","name":"L31 70B Euryale V2.2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}}}},"302ai":{"id":"302ai","env":["302AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.302.ai/v1","name":"302.AI","doc":"https://doc.302.ai","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.145,"output":0.43}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":0,"tiers":[{"input":5,"output":22.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5}}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.286,"output":1.142}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"claude-haiku-4-5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"gemini-3.1-flash-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.72,"output":2.88}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"gpt-5.4-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"gemini-3.5-flash-thinking":{"id":"gemini-3.5-flash-thinking","name":"gemini-3.5-flash-thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"gemini-2.5-flash-image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"gpt-4o":{"id":"gpt-4o","name":"gpt-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25}},"glm-4.6":{"id":"glm-4.6","name":"glm-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"qwen3-235b-a22b-instruct-2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":0.29,"output":1.143}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.16,"output":6.36}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"gpt-5.6-sol-pro":{"id":"gpt-5.6-sol-pro","name":"gpt-5.6-sol-pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.12,"output":0.69}},"claude-sonnet-4-6-thinking":{"id":"claude-sonnet-4-6-thinking","name":"claude-sonnet-4-6-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"glm-5":{"id":"glm-5","name":"glm-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.6}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"gpt-5.2-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"gpt-5.1-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"gemini-2.5-flash-preview-09-2025","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.29,"output":0.86}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5}},"claude-opus-5-thinking":{"id":"claude-opus-5-thinking","name":"claude-opus-5-thinking","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"qwen3-coder-480b-a35b-instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.86,"output":3.43}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"doubao-seed-1-6-vision-250815","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.114,"output":1.143}},"claude-sonnet-4-5-20250929-thinking":{"id":"claude-sonnet-4-5-20250929-thinking","name":"claude-sonnet-4-5-20250929-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.575,"output":2.3}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"grok-4.20-beta-0309-reasoning":{"id":"grok-4.20-beta-0309-reasoning","name":"grok-4.20-beta-0309-reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"gemini-3-pro-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":64000},"cost":{"input":2,"output":120}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"gemini-2.0-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-11","release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":8192},"cost":{"input":0.075,"output":0.3}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.29,"output":2.86}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"gpt-5.6-luna-pro":{"id":"gpt-5.6-luna-pro","name":"gpt-5.6-luna-pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.3}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"qwen3.7-max-2026-06-08":{"id":"qwen3.7-max-2026-06-08","name":"qwen3.7-max-2026-06-08","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.283,"output":1.705}},"kimi-k2-0905-preview":{"id":"kimi-k2-0905-preview","name":"kimi-k2-0905-preview","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.632,"output":2.53}},"deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":10}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.188,"output":1.133}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"doubao-seed-1-6-thinking-250715":{"id":"doubao-seed-1-6-thinking-250715","name":"doubao-seed-1-6-thinking-250715","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16000},"cost":{"input":0.121,"output":1.21}},"gpt-5-thinking":{"id":"gpt-5-thinking","name":"gpt-5-thinking","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax-M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.132,"output":1.254}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"gpt-5.4-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"glm-4.7":{"id":"glm-4.7","name":"glm-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.18,"output":0.564}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"deepseek-v3.2-thinking":{"id":"deepseek-v3.2-thinking","name":"DeepSeek-V3.2-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.29,"output":0.43}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.06,"output":0.46}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":1.08}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"gemini-2.5-flash-nothink":{"id":"gemini-2.5-flash-nothink","name":"gemini-2.5-flash-nothink","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-24","last_updated":"2025-06-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"doubao-seed-1-8-251215":{"id":"doubao-seed-1-8-251215","name":"doubao-seed-1-8-251215","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":224000,"output":64000},"cost":{"input":0.114,"output":0.286}},"claude-opus-4-1-20250805-thinking":{"id":"claude-opus-4-1-20250805-thinking","name":"claude-opus-4-1-20250805-thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-27","last_updated":"2025-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"mistral-large-2512":{"id":"mistral-large-2512","name":"mistral-large-2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":1.1,"output":3.3}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-10-26","last_updated":"2025-10-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.33,"output":1.32}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6}},"gpt-5.6-terra-pro":{"id":"gpt-5.6-terra-pro","name":"gpt-5.6-terra-pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.33,"output":0.33}},"glm-5-turbo":{"id":"glm-5-turbo","name":"glm-5-turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"grok-4.1":{"id":"grok-4.1","name":"grok-4.1","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2,"output":10}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"claude-sonnet-4-6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.29,"output":0.43}},"claude-opus-4-7-thinking":{"id":"claude-opus-4-7-thinking","name":"claude-opus-4-7-thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"claude-opus-4-7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen3-max-2025-09-23":{"id":"qwen3-max-2025-09-23","name":"qwen3-max-2025-09-23","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":258048,"output":65536},"cost":{"input":0.86,"output":3.43}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.285,"output":1.15}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}}}},"openrouter":{"id":"openrouter","env":["OPENROUTER_API_KEY"],"npm":"@openrouter/ai-sdk-provider","api":"https://openrouter.ai/api/v1","name":"OpenRouter","doc":"https://openrouter.ai/models","models":{"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"KAT-Coder-Pro V2.5","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stealth/space-bunny-alpha":{"id":"stealth/space-bunny-alpha","name":"Space Bunny Alpha","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"alpha","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":524288},"cost":{"input":0,"output":0}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3,"tiers":[{"input":0.1,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4,"tiers":[{"input":0.2,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3,"tiers":[{"input":1,"output":6,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.35,"output":11.5,"cache_read":0.3}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Laguna S 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.09,"output":0.18,"cache_read":0.009}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Laguna XS 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.12,"cache_read":0.03}},"prism-ml/ternary-bonsai-2-27b":{"id":"prism-ml/ternary-bonsai-2-27b","name":"Ternary Bonsai 2 27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.5}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex-N2.5-Mini (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex-N2.5-Pro (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"cohere/command-a-plus":{"id":"cohere/command-a-plus","name":"Command A+","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":64000},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"North Mini Code (free)","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943718},"cost":{"input":0.03,"output":0.32,"cache_read":0.016}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.32,"output":0.89}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.462,"output":1.386,"cache_read":0.0154}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":147456},"cost":{"input":0.25,"output":1}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.892272,"output":1.784544,"cache_read":0.074356}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.0763,"output":0.1526,"cache_read":0.01526}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Hy-MT2-30B-A3B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Hy-MT2-1.8B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Hy-MT2-7B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.1875,"output":0.6525}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.08,"cache_read":0.025}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5.3-prime":{"id":"z-ai/glm-5.3-prime","name":"GLM 5.3 Prime","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.8,"output":8.8,"cache_read":0.56}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943718},"cost":{"input":0.15,"output":0.5,"cache_read":0.05}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.09}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":117964},"cost":{"input":0.0605,"output":0.4}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.6496,"output":2.0416,"cache_read":0.12064}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.966,"output":3.036,"cache_read":0.1794}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.2:free":{"id":"z-ai/glm-5.2:free","name":"GLM 5.2 (free)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0,"output":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943717},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Inkling Small (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling:free":{"id":"thinkingmachines/inkling:free","name":"Inkling (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":471859},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"LongCat 2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048756,"output":262144},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"openrouter/free":{"id":"openrouter/free","name":"Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":8000},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":200000}},"openrouter/fusion":{"id":"openrouter/fusion","name":"Fusion","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-11-08","last_updated":"2023-11-08","modalities":{"input":["text","image","audio","pdf","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"GLM Flash Latest","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":128000},"cost":{"input":0.045,"output":0.14,"cache_read":0.01}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"GLM Latest","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":131072},"cost":{"input":0.5614,"output":1.7644,"cache_read":0.10426}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"~openai/gpt-terra-latest":{"id":"~openai/gpt-terra-latest","name":"GPT Terra Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"~openai/gpt-luna-latest":{"id":"~openai/gpt-luna-latest","name":"GPT Luna Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"~openai/gpt-sol-latest":{"id":"~openai/gpt-sol-latest","name":"GPT Sol Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"~openai/gpt-astra-latest":{"id":"~openai/gpt-astra-latest","name":"GPT Astra Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Uncensored","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B ","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.09,"output":0.36,"cache_read":0.018}},"upstage/solar-mini4":{"id":"upstage/solar-mini4","name":"Solar Mini 4","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.005}},"~deepseek/deepseek-flash-latest":{"id":"~deepseek/deepseek-flash-latest","name":"DeepSeek Flash Latest","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.04,"output":1,"cache_read":0.01}},"~deepseek/deepseek-pro-latest":{"id":"~deepseek/deepseek-pro-latest","name":"DeepSeek Pro Latest","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-14","last_updated":"2026-09-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3894,"output":1.1682,"cache_read":0.01239}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":943718},"cost":{"input":0.03,"output":0.32,"cache_read":0.016}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":1.6,"output":4.8,"cache_read":0.4,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LFM2.5-2.6B (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.45,"cache_read":0.04}},"google/gemma-4-31b-it:free":{"id":"google/gemma-4-31b-it:free","name":"Gemma 4 31B (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemma-4-26b-a4b-it:free":{"id":"google/gemma-4-26b-a4b-it:free","name":"Gemma 4 26B A4B (free)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.09,"output":0.3,"cache_read":0.05}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"fireworks/ember-1":{"id":"fireworks/ember-1","name":"Ember-1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-24","last_updated":"2026-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11-30","release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-03-31","release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-01-31","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Fugu Ultra v2","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-08-28","release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"Ling 3.0 Flash Fin (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"Ling 3.0 Flash Sante (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.021,"output":0.063,"cache_read":0.0042}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.6562,"output":3.3,"cache_read":0.18}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"Nemotron 3.5 Content Safety (free)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.08,"output":0.2,"cache_read":0.04}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.08,"output":0.45}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":182520},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"Nemotron 3.5 Lightning (free)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.6-pro-ultraspeed":{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","name":"MiMo-V2.6-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":3686},"cost":{"input":0.08,"output":0.11}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.27,"output":1.08,"cache_read":0.027}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-03-31","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000192,"output":40000},"cost":{"input":0.2,"output":1.1}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.4,"output":2.2}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":176947},"cost":{"input":0.3,"output":1.2}},"mancer/weaver":{"id":"mancer/weaver","name":"Weaver (alpha)","description":"General-purpose chat model for instruction following, writing, and analysis","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots3-Note Preview (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"Aion-3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5":{"id":"aion-labs/aion-3.5","name":"Aion 3.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.5-mini":{"id":"aion-labs/aion-3.5-mini","name":"Aion 3.5 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"Aion-3.0-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.0875,"output":0.35,"cache_read":0.0175}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.475,"output":4.425,"cache_read":0.295,"cache_write":1.84375}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.12,"output":0.8,"cache_read":0.07}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325,"tiers":[{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975}}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"cache_write":0.125,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"cache_write":0.25,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975,"tiers":[{"input":1.56,"output":7.8,"cache_read":0.312,"cache_write":1.95,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.07,"output":0.28}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.52}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"tiers":[{"input":0.78,"output":2.34,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34}}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":1.1,"cache_read":0.07}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375,"tiers":[{"input":0.75,"output":3,"cache_write":0.9375,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":3,"cache_write":0.9375}}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375,"tiers":[{"input":0.325,"output":1.625,"cache_read":0.065,"cache_write":0.40625,"tier":{"type":"context","size":32000}},{"input":0.52,"output":2.6,"cache_read":0.104,"cache_write":0.65,"tier":{"type":"context","size":128000}}]}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"tiers":[{"input":1.56,"output":7.8,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125,"tiers":[{"input":1.17,"output":5.85,"cache_read":0.234,"cache_write":1.4625,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3.8-max-prime":{"id":"qwen/qwen3.8-max-prime","name":"Qwen 3.8 Max Prime","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":4,"output":12,"cache_read":0.5}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375,"tiers":[{"input":0.375,"output":2.25,"cache_write":0.46875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.375,"output":2.25,"cache_write":0.46875}}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375,"tiers":[{"input":1.58,"output":9.48,"cache_write":1.975,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.12,"output":0.5}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.55,"output":3.5,"cache_read":0.225}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":0.3}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56,"tiers":[{"input":0.325,"output":1.95,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.325,"output":1.95}}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262140},"cost":{"input":0.32,"output":2.7,"cache_read":0.15}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625,"tiers":[{"input":1.3,"output":3.9,"cache_write":1.625,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.3,"output":3.9,"cache_write":1.625}}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.12,"output":0.24}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2}}},"qwen/qwen3.8-27b:free":{"id":"qwen/qwen3.8-27b:free","name":"Qwen3.8 27B (free)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.018,"output":0.09}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-10-31","release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-sol-pro":{"id":"openai/gpt-6-sol-pro","name":"GPT-6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"GPT-5 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-06-30","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-6-luna-pro":{"id":"openai/gpt-6-luna-pro","name":"GPT-6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}}}},"perplexity":{"id":"perplexity","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/perplexity","name":"Perplexity","doc":"https://docs.perplexity.ai","models":{"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}}}},"iteracompute":{"id":"iteracompute","env":["ITERACOMPUTE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.iteracompute.com/v1","name":"IteraCompute","doc":"https://iteracompute.com/docs.html","models":{"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":970000,"output":393216},"cost":{"input":0.34,"output":1.05,"cache_read":0.035}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.1,"output":3.3,"cache_read":0.11}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.49,"cache_read":0.03}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":3.5,"cache_read":0.26}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":999999},"cost":{"input":3,"output":14.9,"cache_read":0.29}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":524288},"cost":{"input":0.29,"output":1.2,"cache_read":0.08}},"ornith-ai/ornith-1.5-35b-a3b":{"id":"ornith-ai/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":3,"cache_read":0.03}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":970000,"output":131072},"cost":{"input":1.95,"output":5.95,"cache_read":0.2}}}},"the-grid-ai":{"id":"the-grid-ai","env":["THEGRID_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.thegrid.ai/v1","name":"The Grid AI","doc":"https://thegrid.ai/docs","models":{"agent-prime":{"id":"agent-prime","name":"Agent Prime","description":"Reliable models for dependable agentic applications, multi-step tool use, and reasoning workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"code-max":{"id":"code-max","name":"Code Max","description":"Frontier models for complex research, architectural decisions, debugging, and multi-file development. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"agent-max":{"id":"agent-max","name":"Agent Max","description":"Frontier models for autonomous research, deep multi-step tool chains, and complex long-horizon tasks. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"text-max":{"id":"text-max","name":"Text Max","description":"Frontier models for deep reasoning, long context, and complex workflows. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000}},"code-prime":{"id":"code-prime","name":"Code Prime","description":"Reliable models for everyday software tasks, code completion, review, and standard debugging. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"text-prime":{"id":"text-prime","name":"Text Prime","description":"Reliable models for everyday text generation, editing, and analysis across diverse workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000}},"code-standard":{"id":"code-standard","name":"Code Standard","description":"Price-optimized models for rapid autocomplete, linting, high-frequency suggestions, and batch edits. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"},"text-standard":{"id":"text-standard","name":"Text Standard","description":"Price-optimized models with low-latency, high-throughput and shorter maximum outputs. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000}},"agent-standard":{"id":"agent-standard","name":"Agent Standard","description":"Price-optimized models for fast tool calls, simple agent loops, high-throughput automation, and orchestration. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"}}},"meta":{"id":"meta","env":["META_MODEL_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.meta.ai/v1","name":"Meta","doc":"https://dev.meta.ai/docs","models":{"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}}}},"cline-pass":{"id":"cline-pass","env":["CLINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cline.bot/api/v1","name":"ClinePass","doc":"https://docs.cline.bot/getting-started/clinepass","models":{"cline-pass/mimo-v2.6-pro":{"id":"cline-pass/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"cline-pass/qwen3.7-max":{"id":"cline-pass/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"cline-pass/mimo-v2.5":{"id":"cline-pass/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/glm-5.3-flash":{"id":"cline-pass/glm-5.3-flash","name":"cline-pass/glm-5.3-flash","description":"Latest natively multimodal model in the GLM-5 series","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"cline-pass/qwen3.8-max":{"id":"cline-pass/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"cline-pass/kimi-k3":{"id":"cline-pass/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"cline-pass/deepseek-v4.1-flash":{"id":"cline-pass/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"cline-pass/kimi-k2.6":{"id":"cline-pass/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"cline-pass/mimo-v2.5-pro":{"id":"cline-pass/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}},"cline-pass/mimo-v2.6-flash":{"id":"cline-pass/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/minimax-m3":{"id":"cline-pass/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"cline-pass/glm-5.2":{"id":"cline-pass/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/deepseek-v4-pro":{"id":"cline-pass/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}},"cline-pass/muse-spark-1.3-contributor":{"id":"cline-pass/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072}},"cline-pass/glm-5.3":{"id":"cline-pass/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/kimi-k2.7-code":{"id":"cline-pass/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"cline-pass/qwen3.7-plus":{"id":"cline-pass/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"cline-pass/deepseek-v4-flash":{"id":"cline-pass/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}}}},"modal":{"id":"modal","env":["MODAL_PROXY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.us-west.modal.direct/v1","name":"Modal","doc":"https://modal.com/docs/guide/endpoints","models":{"thinkingmachines/Inkling-NVFP4":{"id":"thinkingmachines/Inkling-NVFP4","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.2,"output":5,"cache_read":0.27}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8-Max","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1010000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"reasoning":15,"cache_read":0.3}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.45,"output":1.5,"cache_read":0.09}}}},"coralbricks":{"id":"coralbricks","env":["CORAL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.coralbricks.ai/v1","name":"CoralBricks","doc":"https://www.coralbricks.ai/docs","models":{"glm-5.3-fp4":{"id":"glm-5.3-fp4","name":"GLM 5.3 FP4","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.12,"output":4.4,"cache_read":0,"cache_write":1.68}},"glm-5.3-flash-fp4":{"id":"glm-5.3-flash-fp4","name":"GLM 5.3 Flash FP4","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0,"cache_write":0.23}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.12,"output":0.6,"cache_read":0,"cache_write":0.18}},"deepseek-v4.1-flash-fast-fp4":{"id":"deepseek-v4.1-flash-fast-fp4","name":"DeepSeek V4.1 Flash FP4","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0,"cache_write":0.09}}}},"routing-run":{"id":"routing-run","env":["ROUTING_RUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.routing.run/v1","name":"routing.run","doc":"https://docs.routing.run/api-reference/models","models":{"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.1,"output":0.1}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.16,"output":0.48}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":0.7,"output":4.2}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":5,"output":25}},"glm-5.2-nitro":{"id":"glm-5.2-nitro","name":"GLM 5.2 Nitro","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"kimi-k2.6-nitro":{"id":"kimi-k2.6-nitro","name":"Kimi K2.6 Nitro","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.348,"output":0.696}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"kimi-k2.7-code-nitro":{"id":"kimi-k2.7-code-nitro","name":"Kimi K2.7 Code Nitro","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":1.5,"output":9}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.112,"output":0.224}}}},"echo":{"id":"echo","env":["ECHO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://echo.tracerml.ai/v1","name":"Echo","doc":"https://echo.tracerml.ai/docs/api","models":{"echo":{"id":"echo","name":"Echo","description":"Adaptive model for coding, reasoning, and tool-driven agent workflows through one OpenAI-compatible endpoint","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"beta","cost":{"input":10,"output":50}}}},"neuralwatt":{"id":"neuralwatt","env":["NEURALWATT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.neuralwatt.com/v1","name":"Neuralwatt","doc":"https://portal.neuralwatt.com/docs","models":{"glm-5.2-short-fast":{"id":"glm-5.2-short-fast","name":"GLM 5.2 Short Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3-flex":{"id":"kimi-k3-flex","name":"Kimi K3 Flex","description":"Kimi K3 on the flex tier: discounted, best-effort latency, requests may be held under load","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.95,"output":9.75,"cache_read":0.195}},"deepseek-v4-flash-flex":{"id":"deepseek-v4-flash-flex","name":"DeepSeek V4 Flash Flex","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.091,"output":0.182,"cache_read":0.0182}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"glm-5.2-short-flex":{"id":"glm-5.2-short-flex","name":"GLM 5.2 Short Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi K3 with thinking disabled for low-latency tool calling, vision, and JSON work","family":"kimi-k3","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3.6-35b-flex":{"id":"qwen3.6-35b-flex","name":"Qwen3.6 35B Flex","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.1885,"output":0.7475,"cache_read":0.01885}},"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.25}},"glm-5.2-short":{"id":"glm-5.2-short","name":"GLM 5.2 Short","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"glm-5.3-flash-flex":{"id":"glm-5.3-flash-flex","name":"GLM-5.3 Flash Flex","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.0975,"output":0.325,"cache_read":0.0195}},"qwen3.6-35b-fast":{"id":"qwen3.6-35b-fast","name":"Qwen3.6 35B Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"glm-5.2-flex":{"id":"glm-5.2-flex","name":"GLM 5.2 Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"deepseek-v4-flash-speed":{"id":"deepseek-v4-flash-speed","name":"DeepSeek V4 Flash (Speed)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"deprecated","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"glm-5.2-short-fast-flex":{"id":"glm-5.2-short-fast-flex","name":"GLM 5.2 Short Fast Flex","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"status":"deprecated","cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"kimi-k2.7-code-fast":{"id":"kimi-k2.7-code-fast","name":"Kimi K2.7 Code Fast","description":"Kimi K2.7 Code with reasoning capped to a short budget for lower latency; reasoning cannot be disabled on this model","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"gemma-4-31b":{"id":"gemma-4-31b","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":16384},"cost":{"input":0.144,"output":0.42,"cache_read":0.0144}},"glm-5.3-flex":{"id":"glm-5.3-flex","name":"GLM 5.3 Flex","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"status":"deprecated","cost":{"input":1,"output":3,"cache_read":0.1}},"qwen-3.8-27b-flex":{"id":"qwen-3.8-27b-flex","name":"Qwen3.8 27B Flex","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":131072},"cost":{"input":0.2925,"output":2.08,"cache_read":0.1625}},"kimi-k2.7-code-flex":{"id":"kimi-k2.7-code-flex","name":"Kimi K2.7 Code Flex","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.6175,"output":2.6,"cache_read":0.06175}},"glm-5.3":{"id":"glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"deepseek-v4.1-flash-flex":{"id":"deepseek-v4.1-flash-flex","name":"DeepSeek V4.1 Flash Flex","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.0975,"output":0.39,"cache_read":0.00975}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}}}},"gitlab":{"id":"gitlab","env":["GITLAB_TOKEN"],"npm":"gitlab-ai-provider","name":"GitLab Duo","doc":"https://docs.gitlab.com/user/duo_agent_platform/","models":{"duo-chat-gpt-5-4-nano":{"id":"duo-chat-gpt-5-4-nano","name":"Agentic Chat (GPT-5.4 Nano)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-opus-5-5":{"id":"duo-chat-opus-5-5","name":"Agentic Chat (Claude Opus 5.5)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-6-astra":{"id":"duo-chat-gpt-6-astra","name":"Agentic Chat (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-5":{"id":"duo-chat-opus-4-5","name":"Agentic Chat (Claude Opus 4.5)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-mini":{"id":"duo-chat-gpt-5-mini","name":"Agentic Chat (GPT-5 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-6-luna":{"id":"duo-chat-gpt-6-luna","name":"Agentic Chat (GPT-6 Luna)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-sonnet-4-5":{"id":"duo-chat-sonnet-4-5","name":"Agentic Chat (Claude Sonnet 4.5)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-6-sol":{"id":"duo-chat-gpt-5-6-sol","name":"Agentic Chat (GPT-5.6 Sol)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-sonnet-4-6":{"id":"duo-chat-sonnet-4-6","name":"Agentic Chat (Claude Sonnet 4.6)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-4-mini":{"id":"duo-chat-gpt-5-4-mini","name":"Agentic Chat (GPT-5.4 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-haiku-4-5":{"id":"duo-chat-haiku-4-5","name":"Agentic Chat (Claude Haiku 4.5)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-fable-5-1":{"id":"duo-chat-fable-5-1","name":"Agentic Chat (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-5":{"id":"duo-chat-gpt-5-5","name":"Agentic Chat (GPT-5.5)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-fable-5":{"id":"duo-chat-fable-5","name":"Agentic Chat (Claude Fable 5)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-3-codex":{"id":"duo-chat-gpt-5-3-codex","name":"Agentic Chat (GPT-5.3 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-1":{"id":"duo-chat-gpt-5-1","name":"Agentic Chat (GPT-5.1)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-6-sol":{"id":"duo-chat-gpt-6-sol","name":"Agentic Chat (GPT-6 Sol)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-7":{"id":"duo-chat-opus-4-7","name":"Agentic Chat (Claude Opus 4.7)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-2":{"id":"duo-chat-gpt-5-2","name":"Agentic Chat (GPT-5.2)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-opus-5":{"id":"duo-chat-opus-5","name":"Agentic Chat (Claude Opus 5)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-6":{"id":"duo-chat-opus-4-6","name":"Agentic Chat (Claude Opus 4.6)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-6-luna":{"id":"duo-chat-gpt-5-6-luna","name":"Agentic Chat (GPT-5.6 Luna)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-codex":{"id":"duo-chat-gpt-5-codex","name":"Agentic Chat (GPT-5 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-sonnet-5":{"id":"duo-chat-sonnet-5","name":"Agentic Chat (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-8":{"id":"duo-chat-opus-4-8","name":"Agentic Chat (Claude Opus 4.8)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-2-codex":{"id":"duo-chat-gpt-5-2-codex","name":"Agentic Chat (GPT-5.2 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-4":{"id":"duo-chat-gpt-5-4","name":"Agentic Chat (GPT-5.4)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-6-terra":{"id":"duo-chat-gpt-5-6-terra","name":"Agentic Chat (GPT-5.6 Terra)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"infer":{"id":"infer","env":["INFER_API_KEY"],"npm":"@ai-sdk/openai","api":"https://infer.flow7.org/v1","name":"Infer by Flow7","doc":"https://infer.flow7.org/opencode","models":{"infer/gpt-6-astra:official":{"id":"infer/gpt-6-astra:official","name":"GPT-6 Astra (Official API)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":271999,"output":128000},"provider":{"shape":"responses"},"cost":{"input":12.5,"output":62.5,"cache_read":1.25,"cache_write":15.625}},"infer/gpt-5.6-sol:official":{"id":"infer/gpt-5.6-sol:official","name":"GPT-5.6 Sol (Official API)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":271999,"output":128000},"provider":{"shape":"responses"},"cost":{"input":2.5,"output":12.5,"cache_read":0.25,"cache_write":3.125}}}},"azure":{"id":"azure","env":["AZURE_RESOURCE_NAME","AZURE_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"gpt-image-2.5-flare":{"id":"gpt-image-2.5-flare","name":"GPT Image 2.5 Flare","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":4096},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":3072},"cost":{"input":0.13,"output":0}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"gpt-image-2.5-sunburst":{"id":"gpt-image-2.5-sunburst","name":"GPT Image 2.5 Sunburst","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-image-1":{"id":"gpt-image-1","name":"GPT-Image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":4096},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek-V4-Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":1.74,"output":3.48}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"GPT-Image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"status":"beta"},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek-V4-Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.19,"output":0.51}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}}}},"freemodel":{"id":"freemodel","env":["FREEMODEL_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://cc.freemodel.dev/v1","name":"FreeModel","doc":"https://freemodel.dev","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}}}},"azure-cognitive-services":{"id":"azure-cognitive-services","env":["AZURE_COGNITIVE_SERVICES_RESOURCE_NAME","AZURE_COGNITIVE_SERVICES_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure Cognitive Services","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":4096},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":4096},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.02,"output":0}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}}}},"pendra":{"id":"pendra","env":["PENDRA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pendra.ai/api/v1","name":"Pendra","doc":"https://pendra.ai/docs/integrations/opencode","models":{"llama3.3:70b":{"id":"llama3.3:70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"qwen3-coder:30b":{"id":"qwen3-coder:30b","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6:27b":{"id":"qwen3.6:27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}}}},"moark":{"id":"moark","env":["MOARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://moark.com/v1","name":"Moark","doc":"https://moark.com/docs/openapi/v1#tag/%E6%96%87%E6%9C%AC%E7%94%9F%E6%88%90","models":{"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":3.5,"output":14}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":2.1,"output":8.4,"cache_read":2.1,"cache_write":8.4}}}},"atomic-chat":{"id":"atomic-chat","env":["ATOMIC_CHAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1337/v1","name":"Atomic Chat","doc":"https://atomic.chat","models":{"gemma-4-E4B-it-MLX-4bit":{"id":"gemma-4-E4B-it-MLX-4bit","name":"Gemma 4 E4B Instruct (MLX 4-bit)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"Meta-Llama-3_1-8B-Instruct-GGUF":{"id":"Meta-Llama-3_1-8B-Instruct-GGUF","name":"Meta Llama 3.1 8B Instruct (GGUF)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0,"output":0}},"Qwen3_5-9B-Q4_K_M":{"id":"Qwen3_5-9B-Q4_K_M","name":"Qwen 3.5 9B (Q4_K_M)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma-4-E4B-it-IQ4_XS":{"id":"gemma-4-E4B-it-IQ4_XS","name":"Gemma 4 E4B Instruct (IQ4_XS)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"Qwen3_5-9B-MLX-4bit":{"id":"Qwen3_5-9B-MLX-4bit","name":"Qwen 3.5 9B (MLX 4-bit)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}}}},"qihang-ai":{"id":"qihang-ai","env":["QIHANG_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qhaigc.net/v1","name":"QiHang","doc":"https://www.qhaigc.net/docs","models":{"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.14,"output":1.14}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.57,"output":3.43}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.04,"output":0.29}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":0.71,"output":3.57}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.71,"tiers":[{"input":0.09,"output":0.71,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.09,"output":0.71}}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.43,"output":2.14}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.07,"output":0.43,"tiers":[{"input":0.07,"output":0.43,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.07,"output":0.43}}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.14,"output":0.71}}}},"ai-router":{"id":"ai-router","env":["AI_ROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ai-router.dev/v1","name":"AI-ROUTER","doc":"https://ai-router.dev/openai-compatible-api-gateway/","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}}}},"llmtr":{"id":"llmtr","env":["LLMTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llmtr.com/v1","name":"LLMTR","doc":"https://llmtr.com/docs","models":{"muse-glimmer-30b-tr":{"id":"muse-glimmer-30b-tr","name":"Muse Glimmer 30B (TR)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"magibu-11b-v8":{"id":"magibu-11b-v8","name":"Magibu 11B v8","description":"Turkish-language chat model for instruction following and assistant flows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.1,"output":0.5}},"gemma-4":{"id":"gemma-4","name":"Gemma 4","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"trendyol-asure-12b":{"id":"trendyol-asure-12b","name":"Trendyol Asure 12B","description":"Turkish-language multimodal instruct model built on Gemma 3 12B for e-commerce text, chat, and image-text tasks","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-19","last_updated":"2026-02-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.1,"output":0.5,"cache_read":0.025}},"medgemma-4b":{"id":"medgemma-4b","name":"MedGemma 4B","description":"Multimodal medical-domain Gemma variant for text and image analysis","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":3,"output":5}},"qwen3-6-35b":{"id":"qwen3-6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":5,"output":10}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"publicai/apertus-70b-instruct":{"id":"publicai/apertus-70b-instruct","name":"Apertus 70B Instruct","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.82,"output":2.92}},"publicai/apertus-8b-instruct":{"id":"publicai/apertus-8b-instruct","name":"Apertus 8B Instruct","description":"Fully open 8B multilingual LLM supporting 1800+ languages with 65K context. Trained on compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.1,"output":0.2}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.58,"output":1.44}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.87,"output":4.68}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.03,"output":0.12}},"upstage/solar-pro3":{"id":"upstage/solar-pro3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro2":{"id":"upstage/solar-pro2","name":"Solar Pro 2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.15,"output":0.6}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.1}},"mimo/mimo-v2.5":{"id":"mimo/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28}},"mimo/mimo-v2.5-pro":{"id":"mimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"cost":{"input":0.2,"output":1.6}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6}}}},"alibaba":{"id":"alibaba","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope-intl.aliyuncs.com/compatible-mode/v1","name":"Alibaba","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.7,"output":2.8}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.8,"output":8.4}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":6}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen-plus-character-ja":{"id":"qwen-plus-character-ja","name":"Qwen Plus Character (Japanese)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":512},"cost":{"input":0.5,"output":1.4}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"qwen3-livetranslate-flash-realtime":{"id":"qwen3-livetranslate-flash-realtime","name":"Qwen3-LiveTranslate Flash Realtime","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":10,"output":10,"input_audio":10,"output_audio":38}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":1.4,"output":5.6}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.27,"output":1.07,"input_audio":4.44,"output_audio":8.89}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"tiers":[{"input":2.7,"output":13.5,"tier":{"type":"context","size":32000}},{"input":4.5,"output":22.5,"tier":{"type":"context","size":128000}}]}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.05,"output":0.2,"reasoning":0.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25,"tiers":[{"input":0.75,"output":3.75,"tier":{"type":"context","size":32000}},{"input":1.2,"output":6,"tier":{"type":"context","size":128000}}]}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.2,"output":4.8}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.175,"output":0.7}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.43,"output":1.66,"input_audio":3.81,"output_audio":15.11}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"reasoning":2.4}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.1,"output":0.4,"input_audio":6.76}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":2.46,"output":7.37}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":5}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.05}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.16,"output":0.49}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.7,"reasoning":2.1}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28,"cache_write":0}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.07,"output":0.27,"input_audio":4.44,"output_audio":8.89}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.52,"output":1.99,"input_audio":4.57,"output_audio":18.13}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.035,"output":0.035}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4,"reasoning":4.2}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-04","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2025-04-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.72,"output":0.72}}}},"auriko":{"id":"auriko","env":["AURIKO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.auriko.ai/v1","name":"Auriko","doc":"https://docs.auriko.ai","models":{"qwen-3.6-plus":{"id":"qwen-3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_write":0.375}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_write":0.375}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}}}},"zenmux":{"id":"zenmux","env":["ZENMUX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://zenmux.ai/api/v1","name":"ZenMux","doc":"https://docs.zenmux.ai","models":{"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3.5-haiku":{"id":"anthropic/claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2024-11-04","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5-free":{"id":"anthropic/claude-sonnet-5-free","name":"Claude Sonnet 5 (Free)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-3.7-sonnet":{"id":"anthropic/claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":4}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek-V3.2 (Non-thinking Mode)","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.28,"output":0.42,"cache_read":0.03}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163000,"output":64000},"cost":{"input":0.22,"output":0.33}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.28,"output":0.43}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":0.572,"cache_read":0.058,"cache_write":0}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.1456,"output":0.4367,"cache_read":0.0291,"tiers":[{"input":0.2911,"output":0.8734,"cache_read":0.0582,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.726,"output":3.1946,"cache_read":0.1743,"tiers":[{"input":1.0165,"output":3.7754,"cache_read":0.2614,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.58,"output":2.6,"cache_read":0.14,"tiers":[{"input":0.87,"output":3.18,"cache_read":0.22,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.6v-flash":{"id":"z-ai/glm-4.6v-flash","name":"GLM 4.6V FlashX","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.0218,"output":0.2184,"cache_read":0.0044,"tiers":[{"input":0.0437,"output":0.4367,"cache_read":0.0044,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3-flashx":{"id":"z-ai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.375,"output":1.25,"cache_read":0.075}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.2911,"output":1.1645,"cache_read":0.0582,"tiers":[{"input":0.5823,"output":2.3291,"cache_read":0.1165,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.7-flash-free":{"id":"z-ai/glm-4.7-flash-free","name":"GLM 4.7 Flash (Free)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.98,"output":3.08,"cache_read":0.182}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.8781,"output":3.5126,"cache_read":0.1903,"tiers":[{"input":1.1709,"output":4.098,"cache_read":0.2927,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.0728,"output":0.4367,"cache_read":0.0146}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.73,"output":3.19,"cache_read":0.174,"tiers":[{"input":1.02,"output":3.77,"cache_read":0.261,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.6v-flash-free":{"id":"z-ai/glm-4.6v-flash-free","name":"GLM 4.6V Flash (Free)","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"tiers":[{"input":0,"output":0,"cache_read":0,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-image":{"id":"z-ai/glm-image","name":"GLM-Image","description":"GLM-Image is an image generation model adopts a hybrid autoregressive + diffusion decoder architecture. In general image generation quality, GLM‑Image aligns with mainstream latent diffusion approaches, but it shows significant advantages in text-rendering and knowledge‑intensive generation scenarios. It performs especially well in tasks requiring precise semantic understanding and complex information expression, while maintaining strong capabilities in high‑fidelity and fine‑grained detail generation. In addition to text‑to‑image generation, GLM‑Image also supports a rich set of image‑to‑image tasks including image editing, style transfer, identity‑preserving generation, and multi‑subject consistency.","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":10240,"output":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.1165,"output":0.2911,"cache_read":0.0233,"tiers":[{"input":0.1747,"output":1.1645,"cache_read":0.0349,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"kuaishou/kat-coder-pro-v2":{"id":"kuaishou/kat-coder-pro-v2","name":"KAT-Coder-Pro-V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"baidu/ernie-5.0-thinking-preview":{"id":"baidu/ernie-5.0-thinking-preview","name":"ERNIE 5.0","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.84,"output":3.37}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"x-ai/grok-voice-tts-1.0":{"id":"x-ai/grok-voice-tts-1.0","name":"Grok Voice TTS 1.0","description":"Convert text into spoken audio with a single API call. The API supports a rich set of expressive voices, inline speech tags for fine-grained delivery control, and output formats from high-fidelity MP3 to telephony-optimized μ-law.","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":15000,"output":15000}},"x-ai/grok-imagine-image-2.0":{"id":"x-ai/grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":66000,"output":0}},"x-ai/grok-voice-stt-1.0":{"id":"x-ai/grok-voice-stt-1.0","name":"Grok Voice STT 1.0","description":"Grok Voice STT 1.0 is xAI's speech-to-text model. It supports transcription with word-level timestamps, optional speaker diarization, and multichannel audio.","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":15000,"output":15000}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.2-fast":{"id":"x-ai/grok-4.2-fast","name":"Grok 4.2 Fast","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":4,"output":12,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"x-ai/grok-4.2-fast-non-reasoning":{"id":"x-ai/grok-4.2-fast-non-reasoning","name":"Grok 4.2 Fast Non Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":4,"output":12,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Doubao-Seed-2.0-mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.03,"output":0.28,"cache_read":0.01,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Doubao-Seed-2.0-lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.09,"output":0.51,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-code":{"id":"volcengine/doubao-seed-code","name":"Doubao-Seed-Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-11","last_updated":"2025-11-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0.17,"output":1.12,"cache_read":0.03}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Doubao-Seed-2.0-pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.45,"output":2.24,"cache_read":0.09,"cache_write":0.0024}},"volcengine/doubao-seed-1.8":{"id":"volcengine/doubao-seed-1.8","name":"Doubao-Seed-1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11,"output":0.28,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.9,"output":4.48}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-19","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.3,"output":2.5,"cache_read":0.07,"cache_write":1}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.03,"cache_write":1}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":65530},"cost":{"input":0.25,"output":1.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"sapiens-ai/agnes-1.5-lite":{"id":"sapiens-ai/agnes-1.5-lite","name":"Agnes 1.5 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.12,"output":0.6}},"sapiens-ai/agnes-1.5-pro":{"id":"sapiens-ai/agnes-1.5-pro","name":"Agnes 1.5 Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-21","last_updated":"2026-03-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.16,"output":0.8}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12-31","release_date":"2026-05-07","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ring-1t":{"id":"inclusionai/ring-1t","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-12","last_updated":"2025-10-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"inclusionai/ling-1t":{"id":"inclusionai/ling-1t","name":"Ling-1T","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262140,"output":262140},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"cost":{"input":0.58,"output":3.02,"cache_read":0.1}},"moonshotai/kimi-k3-free":{"id":"moonshotai/kimi-k3-free","name":"Kimi K3 (Free)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.7-code-free":{"id":"moonshotai/kimi-k2.7-code-free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2-thinking-turbo":{"id":"moonshotai/kimi-k2-thinking-turbo","name":"Kimi K2 Thinking Turbo","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":1.15,"output":8,"cache_read":0.15}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2-omni":{"id":"xiaomi/mimo-v2-omni","name":"MiMo V2 Omni","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":265000,"output":265000},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"xiaomi/mimo-v2-pro":{"id":"xiaomi/mimo-v2-pro","name":"MiMo V2 Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":256000},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax M2.5 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":4.8,"cache_read":0.06,"cache_write":0.75}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3055,"output":1.2219}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":2.4}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.611,"output":2.4439}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"stepfun/step-3":{"id":"stepfun/step-3","name":"Step-3","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":64000},"status":"deprecated","cost":{"input":0.21,"output":0.57}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun/step-3.7-flash-free":{"id":"stepfun/step-3.7-flash-free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1020000,"output":1020000},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.8,"output":4.8}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3-Max-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":1.2,"output":6}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6-Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":3.75,"output":18.75}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":45,"output":225}},"openai/gpt-5.3-chat":{"id":"openai/gpt-5.3-chat","name":"GPT-5.3 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16380},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.2,"output":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-15","last_updated":"2026-01-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":21,"output":168}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.5-instant":{"id":"openai/gpt-5.5-instant","name":"GPT-5.5 Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.75,"output":4.5}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-5.1-chat":{"id":"openai/gpt-5.1-chat","name":"GPT-5.1 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}}}},"unorouter":{"id":"unorouter","env":["UNOROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.unorouter.com/v1","name":"UnoRouter","doc":"https://unorouter.com/models","models":{"gpt-5.5:free":{"id":"gpt-5.5:free","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1.8,"output":10.8}},"glm-4.5-flash:free":{"id":"glm-4.5-flash:free","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.2675,"output":5.3368}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5.4:free":{"id":"gpt-5.4:free","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.819,"output":3.276}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1857,"output":1.1142}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.05,"output":8.4}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.425,"output":2.125}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1875,"output":1.125}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.44,"output":7.2}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6001,"output":5.0288}},"step-3.7-flash:free":{"id":"step-3.7-flash:free","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.2,"output":6}},"deepseek-v4-pro:free":{"id":"deepseek-v4-pro:free","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"glm-5.2:free":{"id":"glm-5.2:free","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.8999,"output":1.7999}},"qwen3.5-397b-a17b:free":{"id":"qwen3.5-397b-a17b:free","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"minimax-m2.7:free":{"id":"minimax-m2.7:free","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0625,"output":0.125}}}},"salad-cloud":{"id":"salad-cloud","env":["SALAD_CLOUD_API_KEY"],"npm":"@saladtechnologies-oss/ai-sdk-provider","name":"SaladCloud AI Gateway","doc":"https://docs.salad.com/ai-gateway/explanation/overview","models":{"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen MoE for agentic tasks, complex reasoning, code generation, and instruction following","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.09,"output":0.6}}}},"vispark":{"id":"vispark","env":["VISPARK_LAB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lab.vispark.in/v1","name":"Vispark","doc":"https://lab.vispark.in/#vision","models":{"vispark/vision-large":{"id":"vispark/vision-large","name":"Vision Large","description":"Most capable Vision model for complex reasoning, detailed media analysis, and structured output over a 1M-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":7.37,"output":22.11}},"vispark/vision-small":{"id":"vispark/vision-small","name":"Vision Small","description":"Fast, low-cost multimodal model for understanding text, images, audio, video, and PDFs, with tool calling and a 1M-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.05,"output":3.16}},"vispark/vision-medium":{"id":"vispark/vision-medium","name":"Vision Medium","description":"Balanced multimodal model pairing a 1M-token context window with deeper reasoning for analysis, content creation, and tool use across text, image, audio, video, and PDF inputs.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-05-15","last_updated":"2026-09","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":4.21,"output":12.63}}}},"siliconflow-cn":{"id":"siliconflow-cn","env":["SILICONFLOW_CN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.cn/v1","name":"SiliconFlow (China)","doc":"https://cloud.siliconflow.com/models","models":{"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3.5-4B":{"id":"Qwen/Qwen3.5-4B","name":"Qwen/Qwen3.5-4B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen/Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.09}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen/Qwen3.6-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen/Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen/Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen/Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.32}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.74}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen/Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.74}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"deepseek-ai/DeepSeek-OCR":{"id":"deepseek-ai/DeepSeek-OCR","name":"deepseek-ai/DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"deepseek-ai/DeepSeek-V4-Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":393000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"Pro/deepseek-ai/DeepSeek-R1":{"id":"Pro/deepseek-ai/DeepSeek-R1","name":"Pro/deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"Pro/deepseek-ai/DeepSeek-V3.2":{"id":"Pro/deepseek-ai/DeepSeek-V3.2","name":"Pro/deepseek-ai/DeepSeek-V3.2","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"Pro/deepseek-ai/DeepSeek-V3":{"id":"Pro/deepseek-ai/DeepSeek-V3","name":"Pro/deepseek-ai/DeepSeek-V3","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"Pro/deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","name":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"Pro/MiniMaxAI/MiniMax-M2.5":{"id":"Pro/MiniMaxAI/MiniMax-M2.5","name":"Pro/MiniMaxAI/MiniMax-M2.5","description":"Frontier MiniMax model for engineering, office tasks, and agentic reasoning","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":131000},"cost":{"input":0.3,"output":1.22}},"Pro/moonshotai/Kimi-K2.6":{"id":"Pro/moonshotai/Kimi-K2.6","name":"Pro/moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"Pro/moonshotai/Kimi-K2.5":{"id":"Pro/moonshotai/Kimi-K2.5","name":"Pro/moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"Pro/zai-org/GLM-5.1":{"id":"Pro/zai-org/GLM-5.1","name":"Pro/zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"Pro/zai-org/GLM-5":{"id":"Pro/zai-org/GLM-5","name":"Pro/zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1,"output":3.2}},"PaddlePaddle/PaddleOCR-VL-1.5":{"id":"PaddlePaddle/PaddleOCR-VL-1.5","name":"PaddlePaddle/PaddleOCR-VL-1.5","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-29","last_updated":"2026-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0,"output":0}}}},"regolo-ai":{"id":"regolo-ai","env":["REGOLO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.regolo.ai/v1","name":"Regolo AI","doc":"https://docs.regolo.ai/","models":{"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":120000},"cost":{"input":0.58,"output":2.42}},"brick-v1-beta":{"id":"brick-v1-beta","name":"Brick v1 Beta","description":"Semantic router by Regolo.ai that directs each request to the most suitable model, optimizing costs and performance","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"status":"beta","cost":{"input":0,"output":0}},"qwen3.5-122b":{"id":"qwen3.5-122b","name":"Qwen3.5-122B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.9,"output":3.6}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3,"output":1.2}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":30000},"cost":{"input":0.46,"output":2.42}},"glm5.2":{"id":"glm5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":96000,"output":96000},"cost":{"input":2.31,"output":6}},"qwen3-reranker-4b":{"id":"qwen3-reranker-4b","name":"Qwen3-Reranker-4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.12,"output":0.12}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":4000},"cost":{"input":0,"output":0}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.7}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":100000},"cost":{"input":0.46,"output":2.42}},"brick-complexity-pro":{"id":"brick-complexity-pro","name":"Brick Complexity Pro","description":"Complexity classifier that powers the Brick semantic router by extracting query difficulty","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"cost":{"input":0.12,"output":0.46}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT-OSS-20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.4,"output":1.8}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.1,"output":0.1}},"mistral-small-4-119b":{"id":"mistral-small-4-119b","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.75,"output":3}},"qwen-image":{"id":"qwen-image","name":"Qwen-Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS-120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1,"output":4.2}},"faster-whisper-large-v3":{"id":"faster-whisper-large-v3","name":"Faster Whisper Large v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0,"output":0}}}},"xiaomi-token-plan-ams":{"id":"xiaomi-token-plan-ams","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-ams.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Europe)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"inceptron":{"id":"inceptron","env":["INCEPTRON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptron.io/v1","name":"Inceptron","doc":"https://docs.inceptron.io","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.13,"output":0.28,"cache_read":0.03,"cache_write":0}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.53,"output":3.39,"cache_read":0.17,"cache_write":0}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.4,"cache_read":0.18,"cache_write":0}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.71,"output":2.35,"cache_read":0.12,"cache_write":0}}}},"upstage":{"id":"upstage","env":["UPSTAGE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.upstage.ai/v1/solar","name":"Upstage","doc":"https://developers.upstage.ai/docs/apis/chat","models":{"solar-pro4":{"id":"solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"solar-pro3":{"id":"solar-pro3","name":"solar-pro3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.25,"output":0.25}},"solar-mini":{"id":"solar-mini","name":"solar-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"solar-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-06-12","last_updated":"2025-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.15,"output":0.15}},"solar-pro2":{"id":"solar-pro2","name":"solar-pro2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.25,"output":0.25}}}},"vultr":{"id":"vultr","env":["VULTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.vultrinference.com/v1","name":"Vultr","doc":"https://api.vultrinference.com/","models":{"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.55,"output":1.65}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":1.2}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":393216,"output":131072},"cost":{"input":0.85,"output":3.1}},"nvidia/DeepSeek-V3.2-NVFP4":{"id":"nvidia/DeepSeek-V3.2-NVFP4","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":1.65}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","name":"NVIDIA Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.38}},"nvidia/Nemotron-Cascade-2-30B-A3B":{"id":"nvidia/Nemotron-Cascade-2-30B-A3B","name":"NVIDIA Nemotron Cascade 2","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}}}},"huggingface":{"id":"huggingface","env":["HF_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://router.huggingface.co/v1","name":"Hugging Face","doc":"https://huggingface.co/docs/inference-providers","models":{"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"tencent/Hy4-preview":{"id":"tencent/Hy4-preview","name":"Hy4 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.834,"output":2.501}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.06,"output":0.06}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.59,"output":0.79}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.3}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"MiMo model for long-context reasoning, perception, and agentic tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.4,"output":2}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.5,"output":1.2}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.25,"output":1}},"Qwen/Qwen2.5-Coder-32B-Instruct":{"id":"Qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen2.5-Coder-32B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.06,"output":0.2}},"Qwen/Qwen3-Coder-Next":{"id":"Qwen/Qwen3-Coder-Next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"Qwen/Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen/Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next-80B-A3B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.95}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":3}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.855,"output":2.565}},"Qwen/Qwen3-235B-A22B":{"id":"Qwen/Qwen3-235B-A22B","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.29,"output":0.59}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.07,"output":0.26}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen 3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.01,"output":0}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3.6}},"Qwen/Qwen3-Embedding-4B":{"id":"Qwen/Qwen3-Embedding-4B","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":6.25}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":2,"output":2}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.47,"output":3.19}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32768},"cost":{"input":0.7,"output":2.5}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.28,"output":0.4}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":3,"output":5}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":8192},"cost":{"input":0.4,"output":1.3}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32}},"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.1,"output":0.3}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-10","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2":{"id":"MiniMaxAI/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15}},"moonshotai/Kimi-K2-Instruct":{"id":"moonshotai/Kimi-K2-Instruct","name":"Kimi-K2-Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-14","last_updated":"2025-07-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":3}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi-K2-Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi-K2-Instruct-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1,"output":3}},"zai-org/GLM-4.6V-Flash":{"id":"zai-org/GLM-4.6V-Flash","name":"GLM-4.6V-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"zai-org/GLM-4.5V":{"id":"zai-org/GLM-4.5V","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":0.69}}}},"volcengine":{"id":"volcengine","env":["ARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/v3","name":"Volcengine Ark","doc":"https://www.volcengine.com/docs/82379/1330310","models":{"glm-5-3-flash-260828":{"id":"glm-5-3-flash-260828","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11875,"output":0.41563,"cache_read":0.03414}},"doubao-seed-1-6-251015":{"id":"doubao-seed-1-6-251015","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.08906,"output":0.53436,"cache_read":0.01781,"tiers":[{"input":0.13359,"output":0.80154,"cache_read":0.02672,"tier":{"type":"context","size":32000}},{"input":0.26718,"output":1.60308,"cache_read":0.05344,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-1-pro-260628":{"id":"doubao-seed-2-1-pro-260628","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.02969,"output":0.29687,"cache_read":0.00594,"tiers":[{"input":0.05937,"output":0.59374,"cache_read":0.01187,"tier":{"type":"context","size":32000}},{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-1-turbo-260628":{"id":"doubao-seed-2-1-turbo-260628","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.4453,"output":2.22651,"cache_read":0.08906}},"glm-5-2-260617":{"id":"glm-5-2-260617","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.18747,"output":4.15615,"cache_read":0.29687}},"doubao-seed-1-8-251228":{"id":"doubao-seed-1-8-251228","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-1-6-flash-250828":{"id":"doubao-seed-1-6-flash-250828","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.02227,"output":0.22265,"cache_read":0.00445,"tiers":[{"input":0.04453,"output":0.4453,"cache_read":0.00445,"tier":{"type":"context","size":32000}},{"input":0.08906,"output":0.8906,"cache_read":0.00445,"tier":{"type":"context","size":128000}}]}},"doubao-seed-character-260628":{"id":"doubao-seed-character-260628","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.11875,"output":0.29687,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":0.8906,"cache_read":0.02375,"tier":{"type":"context","size":32000}}]}},"deepseek-v4-pro-ga-260813":{"id":"deepseek-v4-pro-ga-260813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.3359,"output":4.00771,"cache_read":0.04453}},"deepseek-v4-flash-ga-260731":{"id":"deepseek-v4-flash-ga-260731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4453,"output":1.3359,"cache_read":0.01484}}}},"impossibl":{"id":"impossibl","env":["IMPOSSIBL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.impossibl.com/v1","name":"Impossibl","doc":"https://impossibl.com/docs/models","models":{"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.19,"output":0.51,"cache_read":0.028}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"fireworks/gpt-oss-20b":{"id":"fireworks/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.3,"cache_read":0.035}},"fireworks/glm-5.2":{"id":"fireworks/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"fireworks/gpt-oss-120b":{"id":"fireworks/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"qwen/qwen3.8-max-preview":{"id":"qwen/qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"tiers":[{"input":1,"output":4,"cache_read":0.2,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1,"output":4,"cache_read":0.2}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}}}},"xpersona":{"id":"xpersona","env":["XPERSONA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://www.xpersona.co/v1","name":"Xpersona","doc":"https://www.xpersona.co/docs","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.75,"output":6,"reasoning":6,"cache_read":0.075}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3.7,"reasoning":3.7,"cache_read":0.06}},"xpersona-frieren-coder":{"id":"xpersona-frieren-coder","name":"Xpersona Frieren 1","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-01","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":1.5,"output":6,"reasoning":6,"cache_read":0.15}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"xpersona-gpt-5.5":{"id":"xpersona-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18,"reasoning":18,"cache_read":0.3}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.55,"output":12.2,"reasoning":12.2,"cache_read":0.155}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18.5,"reasoning":18.5,"cache_read":0.3}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.375,"output":4,"reasoning":4,"cache_read":0.0375}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.5,"output":9.25,"reasoning":9.25,"cache_read":0.15}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":5.55,"reasoning":5.55,"cache_read":0.09}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":2,"reasoning":2,"cache_read":0.15}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}}}},"qiniu-ai":{"id":"qiniu-ai","env":["QINIU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qnaigc.com/v1","name":"Qiniu","doc":"https://developer.qiniu.com/aitokenapi","models":{"glm-4.5":{"id":"glm-4.5","name":"GLM 4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304}},"gemini-3.0-flash-preview":{"id":"gemini-3.0-flash-preview","name":"Gemini 3.0 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"doubao-seed-2.0-mini":{"id":"doubao-seed-2.0-mini","name":"Doubao Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Doubao Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"doubao-seed-1.6-thinking":{"id":"doubao-seed-1.6-thinking","name":"Doubao-Seed 1.6 Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"gemini-3.0-pro-image-preview":{"id":"gemini-3.0-pro-image-preview","name":"Gemini 3.0 Pro Image Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235b A22B Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"qwen-max-2025-01-25":{"id":"qwen-max-2025-01-25","name":"Qwen2.5-Max-2025-01-25","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek-V3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-14","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":4096}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen-Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":4096}},"gemini-2.0-flash":{"id":"gemini-2.0-flash","name":"Gemini 2.0 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"Gemini 2.0 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"claude-4.0-sonnet":{"id":"claude-4.0-sonnet","name":"Claude 4.0 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"qwen2.5-vl-7b-instruct":{"id":"qwen2.5-vl-7b-instruct","name":"Qwen 2.5 VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":4096}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen 3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"doubao-seed-2.0-pro":{"id":"doubao-seed-2.0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"doubao-seed-1.6-flash":{"id":"doubao-seed-1.6-flash","name":"Doubao-Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-06","last_updated":"2025-09-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"qwen3-vl-30b-a3b-thinking":{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen3-Vl 30b A3b Thinking","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":12000}},"claude-3.5-sonnet":{"id":"claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8200}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"qwen-vl-max-2025-01-25":{"id":"qwen-vl-max-2025-01-25","name":"Qwen VL-MAX-2025-01-25","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"doubao-1.5-thinking-pro":{"id":"doubao-1.5-thinking-pro","name":"Doubao 1.5 Thinking Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"kling-v2-6":{"id":"kling-v2-6","name":"Kling-V2 6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":99999999,"output":99999999}},"claude-4.1-opus":{"id":"claude-4.1-opus","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":80000}},"doubao-seed-1.6":{"id":"doubao-seed-1.6","name":"Doubao-Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-22","last_updated":"2026-02-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"doubao-seed-2.0-code":{"id":"doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"claude-4.0-opus":{"id":"claude-4.0-opus","name":"Claude 4.0 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen3 30b A3b Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen 2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"doubao-1.5-vision-pro":{"id":"doubao-1.5-vision-pro","name":"Doubao 1.5 Vision Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"deepseek-v3.1":{"id":"deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen3-30b-a3b-thinking-2507":{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen3 30b A3b Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":126000,"output":32000}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":4096}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"gemini-3.0-pro-preview":{"id":"gemini-3.0-pro-preview","name":"Gemini 3.0 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"deepseek/deepseek-v3.2-exp-thinking":{"id":"deepseek/deepseek-v3.2-exp-thinking","name":"DeepSeek/DeepSeek-V3.2-Exp-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-math-v2":{"id":"deepseek/deepseek-math-v2","name":"Deepseek/Deepseek-Math-V2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":160000,"output":160000}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-251201":{"id":"deepseek/deepseek-v3.2-251201","name":"Deepseek/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.1-terminus-thinking":{"id":"deepseek/deepseek-v3.1-terminus-thinking","name":"DeepSeek/DeepSeek-V3.1-Terminus-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"Z-AI/GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"Z-Ai/GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"Z-Ai/GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/autoglm-phone-9b":{"id":"z-ai/autoglm-phone-9b","name":"Z-Ai/Autoglm Phone 9b","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":12800,"output":4096}},"meituan/longcat-flash-chat":{"id":"meituan/longcat-flash-chat","name":"Meituan/Longcat-Flash-Chat","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-05","last_updated":"2025-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"meituan/longcat-flash-lite":{"id":"meituan/longcat-flash-lite","name":"Meituan/Longcat-Flash-Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":320000}},"x-ai/grok-4-fast-reasoning":{"id":"x-ai/grok-4-fast-reasoning","name":"X-Ai/Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-reasoning":{"id":"x-ai/grok-4.1-fast-reasoning","name":"X-Ai/Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":20000000,"output":2000000}},"x-ai/grok-4-fast-non-reasoning":{"id":"x-ai/grok-4-fast-non-reasoning","name":"X-Ai/Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-code-fast-1":{"id":"x-ai/grok-code-fast-1","name":"x-AI/Grok-Code-Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"x-AI/Grok-4.1-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-non-reasoning":{"id":"x-ai/grok-4.1-fast-non-reasoning","name":"X-Ai/Grok 4.1 Fast Non Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4-fast":{"id":"x-ai/grok-4-fast","name":"x-AI/Grok-4-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-20","last_updated":"2025-09-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"stepfun-ai/gelab-zero-4b-preview":{"id":"stepfun-ai/gelab-zero-4b-preview","name":"Stepfun-Ai/Gelab Zero 4b Preview","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"Xiaomi/Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"Minimax/Minimax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"Minimax/Minimax-M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax/Minimax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"Minimax/Minimax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Stepfun/Step-3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"output":4096}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"OpenAI/GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"OpenAI/GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}}}},"modelscope":{"id":"modelscope","env":["MODELSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-inference.modelscope.cn/v1","name":"ModelScope","doc":"https://modelscope.cn/docs/model-service/API-Inference/intro","models":{"ZhipuAI/GLM-4.6":{"id":"ZhipuAI/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":98304},"cost":{"input":0,"output":0}},"ZhipuAI/GLM-4.5":{"id":"ZhipuAI/GLM-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Thinking-2507":{"id":"Qwen/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}}}},"google":{"id":"google","env":["GOOGLE_API_KEY","GOOGLE_GENERATIVE_AI_API_KEY","GEMINI_API_KEY"],"npm":"@ai-sdk/google","name":"Google","doc":"https://ai.google.dev/gemini-api/docs/models","models":{"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-omni-flash-preview":{"id":"gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Video generation and editing model for fast, conversational text- and image-to-video workflows","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1.5,"output":17.5}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":60}},"veo-3.1-lite-generate-preview":{"id":"veo-3.1-lite-generate-preview","name":"Veo 3.1 lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"lyria-3-pro-preview":{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Music generation model for full-length songs from text or images with vocals and structure","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-3.1-flash-tts-preview":{"id":"gemini-3.1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30,"cache_read":0.075}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1},"cost":{"input":0.2,"output":0,"input_audio":6.5}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":120}},"gemini-2.5-pro-preview-tts":{"id":"gemini-2.5-pro-preview-tts","name":"Gemini 2.5 Pro Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60}},"lyria-3-clip-preview":{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Music generation model for short 30-second clips, loops, and previews from text or image prompts","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-computer-use-preview-10-2025":{"id":"gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview 10-2025","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":1.25,"output":10,"tiers":[{"input":2.5,"output":15,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15}}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":120}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"gemini-2.5-flash-preview-tts":{"id":"gemini-2.5-flash-preview-tts","name":"Gemini 2.5 Flash Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-3.1-flash-lite-image":{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":30}},"deep-research-preview-04-2026":{"id":"deep-research-preview-04-2026","name":"Deep Research Preview (Apr-21-2026)","description":"Agentic model for autonomous multi-step research, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"deep-research-max-preview-04-2026":{"id":"deep-research-max-preview-04-2026","name":"Deep Research Max Preview (Apr-21-2026)","description":"Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"veo-3.1-fast-generate-preview":{"id":"veo-3.1-fast-generate-preview","name":"Veo 3.1 fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"veo-3.1-generate-preview":{"id":"veo-3.1-generate-preview","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192},"status":"beta"},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.5-live-translate-preview":{"id":"gemini-3.5-live-translate-preview","name":"Gemini 3.5 Live Translate Preview","description":"Low-latency audio-to-audio model for real-time speech translation across 70+ languages","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["audio"],"output":["audio","text"]},"open_weights":false,"limit":{"context":16384,"output":32768},"cost":{"input":3.5,"output":21,"input_audio":3.5,"output_audio":21}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Legacy model retained for compatibility with older integrations","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gemini-3.1-flash-live-preview":{"id":"gemini-3.1-flash-live-preview","name":"Gemini 3.1 Flash Live Preview","description":"High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image","video","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.75,"output":4.5,"input_audio":3,"output_audio":12}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}}}},"vancine":{"id":"vancine","env":["VANCINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://vancine.com/v1","name":"Vancine","doc":"https://vancine.com/docs","models":{"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.24,"output":0.96,"cache_read":0.048}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.4,"cache_read":0.024}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6,"output":4.8,"cache_read":0.2}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.4,"output":12,"cache_read":0.24}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.24,"output":0.96,"cache_read":0.0048}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.013}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.67,"output":2,"cache_read":0.034}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.12,"output":3.52,"cache_read":0.208}}}},"zhipuai-coding-plan":{"id":"zhipuai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/coding/paas/v4","name":"Zhipu AI Coding Plan","doc":"https://docs.bigmodel.cn/cn/coding-plan/overview","models":{"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}}}},"lucidquery":{"id":"lucidquery","env":["LUCIDQUERY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lucidquery.com/v1","name":"LucidQuery","doc":"https://lucidquery.com/docs","models":{"lucidquery-agi-01-frontier":{"id":"lucidquery-agi-01-frontier","name":"AGI-01 Frontier","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":4.5,"output":22}},"lucidquery-nexus-coder":{"id":"lucidquery-nexus-coder","name":"LucidQuery Nexus Coder","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"lucid","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-01","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":250000,"output":60000},"cost":{"input":2,"output":5}},"lucidnova-rf1-100b":{"id":"lucidnova-rf1-100b","name":"LucidNova RF1 100B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-09-16","release_date":"2024-12-28","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":120000,"output":8000},"cost":{"input":2,"output":5}},"lucidquery-agi-01-swift":{"id":"lucidquery-agi-01-swift","name":"AGI-01 Swift","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":2.5,"output":15}}}},"gmicloud":{"id":"gmicloud","env":["GMICLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.gmi-serving.com/v1","name":"GMI Cloud","doc":"https://docs.gmicloud.ai/inference-engine/api-reference/llm-api-reference","models":{"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":384000},"cost":{"input":0.112,"output":0.224,"cache_read":0.022}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.392,"output":2.784,"cache_read":0.116}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.855,"output":3.6,"cache_read":0.144}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.979,"output":3.08,"cache_read":0.182}},"zai-org/GLM-5-FP8":{"id":"zai-org/GLM-5-FP8","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"oci":{"id":"oci","env":["OCI_GENAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.generativeai.us-chicago-1.oci.oraclecloud.com/openai/v1","name":"OCI Generative AI","doc":"https://docs.oracle.com/en-us/iaas/Content/generative-ai/pretrained-models.htm","models":{"xai.grok-4.3":{"id":"xai.grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai.grok-4.20-non-reasoning":{"id":"xai.grok-4.20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"meta.llama-3.3-70b-instruct":{"id":"meta.llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"meta.llama-4-scout-17b-16e-instruct":{"id":"meta.llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":192000,"output":16384},"cost":{"input":0.72,"output":0.72}},"xai.grok-4.6":{"id":"xai.grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"meta.llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta.llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":16384},"cost":{"input":0.72,"output":0.72}},"openai.gpt-oss-120b":{"id":"openai.gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"openai.gpt-oss-20b":{"id":"openai.gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"xai.grok-4.20-reasoning":{"id":"xai.grok-4.20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}}}},"cloudflare-ai-gateway":{"id":"cloudflare-ai-gateway","env":["CLOUDFLARE_API_TOKEN","CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_GATEWAY_ID"],"npm":"ai-gateway-provider","name":"Cloudflare AI Gateway","doc":"https://developers.cloudflare.com/ai-gateway/","models":{"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"unbiased/pareto":{"id":"unbiased/pareto","name":"Pareto","description":"Blended multimodal model that runs multiple language models in parallel and returns a synthesized answer","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"alibaba/qwen3.5-397b-a17b":{"id":"alibaba/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064}},"xai/grok-4.7":{"id":"xai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":5,"cache_read":0.625}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}}}},"clarifai":{"id":"clarifai","env":["CLARIFAI_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://api.clarifai.com/v2/ext/openai/v1","name":"Clarifai","doc":"https://docs.clarifai.com/compute/inference/","models":{"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput":{"id":"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput","name":"MiniMax-M2.5 High Throughput","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"clarifai/main/models/mm-poly-8b":{"id":"clarifai/main/models/mm-poly-8b","name":"MM Poly 8B","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"mm-poly","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.658,"output":1.11}},"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR":{"id":"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR","name":"DeepSeek OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"deepseek","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.2,"output":0.7}},"mistralai/completion/models/Ministral-3-3B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-3B-Reasoning-2512","name":"Ministral 3 3B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.039,"output":0.54825}},"mistralai/completion/models/Ministral-3-14B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-14B-Reasoning-2512","name":"Ministral 3 14B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-01","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":1.7}},"moonshotai/chat-completion/models/Kimi-K2_6":{"id":"moonshotai/chat-completion/models/Kimi-K2_6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"arcee_ai/AFM/models/trinity-mini":{"id":"arcee_ai/AFM/models/trinity-mini","name":"Trinity Mini","description":"Reasoning-tuned 26B MoE model with 3B active parameters for agents, tools, and multi-step workloads","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-01","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.045,"output":0.15}},"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct":{"id":"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.11458,"output":0.74812}},"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.5}},"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.36,"output":1.3}},"openai/chat-completion/models/gpt-oss-120b-high-throughput":{"id":"openai/chat-completion/models/gpt-oss-120b-high-throughput","name":"GPT OSS 120B High Throughput","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.36}},"openai/chat-completion/models/gpt-oss-20b":{"id":"openai/chat-completion/models/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.045,"output":0.18}}}},"aiand":{"id":"aiand","env":["AIAND_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aiand.com/v1","name":"ai&","doc":"https://docs.aiand.com/","models":{"motif-technologies/motif-3":{"id":"motif-technologies/motif-3","name":"Motif 3","description":"Motif 3 is a large-scale, decoder-only Mixture-of-Experts (MoE) language model with 314 billion total parameters and 13.2 billion parameters activated per token.","family":"motif","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2,"cache_read":0.2}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1,"output":2.5,"cache_read":0.25}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.25,"cache_read":0.08}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":12.5,"cache_read":0.5}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5,"cache_read":0.2}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4,"cache_read":0.3}},"zai-org/glm-5.3":{"id":"zai-org/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4,"cache_read":0.3}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3,"cache_read":0.2}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2,"cache_read":0.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.08}}}},"frogbot":{"id":"frogbot","env":["FROGBOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://app.frogbot.ai/api/v1","name":"FrogBot","doc":"https://docs.frogbot.ai","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"minimax-m2-5":{"id":"minimax-m2-5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-01-15","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.2}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-07-17","last_updated":"2025-07-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek v4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":1.74,"output":3.48,"cache_read":0.14}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"zai-glm-5-1":{"id":"zai-glm-5-1","name":"Z.AI GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-20","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":8192},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"stackit":{"id":"stackit","env":["STACKIT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1","name":"STACKIT","doc":"https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models","models":{"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-05-17","last_updated":"2025-05-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":37000,"output":4096},"cost":{"input":0.53,"output":0.76}},"intfloat/e5-mistral-7b-instruct":{"id":"intfloat/e5-mistral-7b-instruct","name":"E5 Mistral 7B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.02,"output":0.02}},"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8","name":"Qwen3-VL 235B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":218000,"output":16384},"cost":{"input":1.76,"output":2.05}},"Qwen/Qwen3-VL-Embedding-8B":{"id":"Qwen/Qwen3-VL-Embedding-8B","name":"Qwen3-VL Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.09,"output":0.09}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.53,"output":0.76}},"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic":{"id":"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.53,"output":0.76}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.29}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":8192},"cost":{"input":0.53,"output":0.76}}}},"anyapi":{"id":"anyapi","env":["ANYAPI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.anyapi.ai/v1","name":"AnyAPI","doc":"https://docs.anyapi.ai","models":{"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000}},"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated"},"mistralai/mistral-large-2512":{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}}}},"crusoe":{"id":"crusoe","env":["CRUSOE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.crusoecloud.com/v1","name":"Crusoe","doc":"https://docs.crusoecloud.com/managed-inference/overview","models":{"zai/GLM-5.1":{"id":"zai/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4.4,"cache_read":0.25}},"zai/GLM-5.2":{"id":"zai/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.25,"output":0.75,"cache_read":0.13}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4,"cache_read":0.14}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.8,"cache_read":0.11}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.35}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.3,"output":1.83,"cache_read":0.3,"input_audio":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2,"cache_read":0.05}}}},"volcengine-coding-plan":{"id":"volcengine-coding-plan","env":["ARK_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/coding/v3","name":"Volcengine Ark Coding Plan","doc":"https://www.volcengine.com/docs/82379/1928261","models":{"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-2.1-turbo":{"id":"doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}}}},"jiekou":{"id":"jiekou","env":["JIEKOU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jiekou.ai/openai","name":"Jiekou.AI","doc":"https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"grok-4-1-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"gpt-5.2-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"gpt-5.1-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":10.8}},"gpt-5-codex":{"id":"gpt-5-codex","name":"gpt-5-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"gpt-5.2-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":18.9,"output":151.2}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":1.1,"output":4.4}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"gemini-2.5-flash-lite-preview-06-17","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","video","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"claude-opus-4-20250514","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":13.5,"output":108}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"gpt-5-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":4.5,"output":22.5}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.27,"output":2.25}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"gemini-2.5-flash-preview-05-20","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":0.135,"output":3.15}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"gpt-5.1-codex-max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.575,"output":12.6}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"grok-code-fast-1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.18,"output":1.35}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"claude-opus-4-6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":64000},"cost":{"input":0.9,"output":4.5}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.36}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"gemini-2.5-pro-preview-06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"claude-sonnet-4-20250514","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"gpt-5.1-codex-mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"o3":{"id":"o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":10,"output":40}},"grok-4-0709":{"id":"grok-4-0709","name":"grok-4-0709","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.7,"output":13.5}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.28,"output":1.14}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32767}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":262143}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":131071}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.15,"output":0.8}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"qwen/qwen3-coder-next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.2}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":3}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}}}},"ollama-cloud":{"id":"ollama-cloud","env":["OLLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ollama.com/v1","name":"Ollama Cloud","doc":"https://docs.ollama.com/cloud","models":{"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"nemotron-3-ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.1,"output":3,"cache_read":0.1}},"kimi-k3":{"id":"kimi-k3","name":"kimi-k3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"minimax-m2.5":{"id":"minimax-m2.5","name":"minimax-m2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072}},"kimi-k2.6":{"id":"kimi-k2.6","name":"kimi-k2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"qwen3.5:397b":{"id":"qwen3.5:397b","name":"qwen3.5:397b","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"release_date":"2026-02-15","last_updated":"2026-02-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"gpt-oss:20b":{"id":"gpt-oss:20b","name":"gpt-oss:20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.3,"cache_read":0.035}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"gpt-oss:120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.014}},"nemotron-3-nano:30b":{"id":"nemotron-3-nano:30b","name":"nemotron-3-nano:30b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.06,"output":0.24}},"kimi-k2.5":{"id":"kimi-k2.5","name":"kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"mistral-large-3:675b":{"id":"mistral-large-3:675b","name":"mistral-large-3:675b","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-12-02","last_updated":"2026-01-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"minimax-m2.7":{"id":"minimax-m2.7","name":"minimax-m2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"deepseek-v4-pro:0813":{"id":"deepseek-v4-pro:0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"minimax-m3":{"id":"minimax-m3","name":"minimax-m3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"deepseek-v4-flash:0731":{"id":"deepseek-v4-flash:0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"gemma4:31b":{"id":"gemma4:31b","name":"gemma4:31b","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.4,"cache_read":0.05}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":976000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"glm-5.1":{"id":"glm-5.1","name":"glm-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-03-27","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"deepseek-v4-pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"kimi-k2.7-code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"nemotron-3-super":{"id":"nemotron-3-super","name":"nemotron-3-super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.015,"output":0.6,"cache_read":0.015}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}}}},"agentrouter":{"id":"agentrouter","env":["AGENTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://agentrouter.org/v1","name":"AgentRouter","doc":"https://agentrouter.org/docs/opencode.html","models":{"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}}}},"tencent-tokenhub":{"id":"tencent-tokenhub","env":["TENCENT_TOKENHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://tokenhub.tencentmaas.com/v1","name":"Tencent TokenHub","doc":"https://cloud.tencent.com/document/product/1823/130050","models":{"hy3-preview":{"id":"hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}}}},"watsonx":{"id":"watsonx","env":["WATSONX_AI_APIKEY","WATSONX_AI_PROJECT_ID"],"npm":"watsonx-ai-provider","name":"watsonx.ai","doc":"https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models","models":{"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.371,"output":1.484}},"meta-llama/llama-3-3-70b-instruct":{"id":"meta-llama/llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.7526,"output":0.7526}},"ibm/granite-4-h-small":{"id":"ibm/granite-4-h-small","name":"Granite-4.0-H-Small","description":"Open-weight hybrid model for enterprise chat, coding, retrieval-augmented generation, and tool-calling workloads","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0636,"output":0.265}},"mistralai/mistral-small-3-1-24b-instruct-2503":{"id":"mistralai/mistral-small-3-1-24b-instruct-2503","name":"Mistral Small 3.1 24B","description":"Efficient multimodal model for instruction following, coding, reasoning, and function calling","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.106,"output":0.318}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.159,"output":0.636}}}},"ambient":{"id":"ambient","env":["AMBIENT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ambient.xyz/v1","name":"Ambient","doc":"https://ambient.xyz","models":{"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.08,"output":0.18,"cache_read":0.016,"cache_write":0}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.028,"cache_write":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}},"ambient/large":{"id":"ambient/large","name":"Ambient Large","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.2,"cache_write":0}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.69,"output":3.49,"cache_read":0.14,"cache_write":0}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.2,"output":4.2,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"cache_write":0}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.19,"output":1.14,"cache_read":0.03,"cache_write":0}}}},"model-oracle-ai":{"id":"model-oracle-ai","env":["MODEL_ORACLE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.modeloracle.com/api/v1","name":"Model Oracle AI","doc":"https://modeloracle.com/setup/","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"auto":{"id":"auto","name":"Auto","description":"Model Oracle AI decision engine that selects and routes among configured coding-agent models","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-29","last_updated":"2026-07-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}}}},"xai":{"id":"xai","env":["XAI_API_KEY"],"npm":"@ai-sdk/xai","name":"xAI","doc":"https://docs.x.ai/docs/models","models":{"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"grok-imagine-image":{"id":"grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-imagine-video":{"id":"grok-imagine-video","name":"Grok Imagine Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.20-0309-reasoning":{"id":"grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-imagine-video-1.5":{"id":"grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Video model for image-to-video generation, editing, and extension workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text","image","audio","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"grok-4.20-0309-non-reasoning":{"id":"grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-imagine-image-quality":{"id":"grok-imagine-image-quality","name":"Grok Imagine Image Quality","description":"Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"grok-4.20-multi-agent-0309":{"id":"grok-4.20-multi-agent-0309","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}}}},"nebius":{"id":"nebius","env":["NEBIUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenfactory.nebius.com/v1","name":"Nebius Token Factory","doc":"https://docs.tokenfactory.nebius.com/","models":{"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma-3-27b-it","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-10","release_date":"2026-01-20","last_updated":"2026-02-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":110000,"input":100000,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-28","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-10","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"input":40960,"output":0},"cost":{"input":0.01,"output":0}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2026-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":250000,"output":8192},"cost":{"input":0.6,"output":3.6,"cache_read":0.06,"cache_write":0.75}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":979000,"output":979000},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":1048000},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.15}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8000},"cost":{"input":3,"output":15,"cache_read":3}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8000},"cost":{"input":0.95,"output":4}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.15,"output":0.5,"cache_read":0.15}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron-3-Super-120B-A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.3,"output":0.9}},"nvidia/Nemotron-3_5-Lightning":{"id":"nvidia/Nemotron-3_5-Lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":3,"cache_read":1}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":124000,"output":8192},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.015,"cache_write":0.18}},"NousResearch/Hermes-4-405B":{"id":"NousResearch/Hermes-4-405B","name":"Hermes-4-405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-01-30","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":120000,"output":8192},"cost":{"input":1,"output":3,"reasoning":3,"cache_read":0.1,"cache_write":1.25}}}},"minimax-cn":{"id":"minimax-cn","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.cn/anthropic/v1","name":"MiniMax (minimax.cn)","doc":"https://platform.minimaxi.com/docs/guides/quickstart","models":{"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}}}},"scaleway":{"id":"scaleway","env":["SCALEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scaleway.ai/v1","name":"Scaleway","doc":"https://www.scaleway.com/en/docs/generative-apis/","models":{"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.468,"output":0.936,"reasoning":0.936,"cache_read":0.0936}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":16384},"cost":{"input":0.75,"output":2.25,"reasoning":8.4}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2026-03-17","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":8192},"cost":{"input":0.003,"output":0}},"bge-multilingual-gemma2":{"id":"bge-multilingual-gemma2","name":"BGE Multilingual Gemma2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-26","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.1,"output":0}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.8}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.25,"output":0.5}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":0.9,"output":0.9}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":0.25,"output":1.5}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B 2409","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-25","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.1,"output":0}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5 128B","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.5,"output":7.5}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.6,"output":3.6}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.8,"output":5.5}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2 24B Instruct (2506)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.35}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.6}}}},"vercel":{"id":"vercel","env":["AI_GATEWAY_API_KEY"],"npm":"@ai-sdk/gateway","name":"Vercel AI Gateway","doc":"https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway","models":{"voyage/voyage-law-2":{"id":"voyage/voyage-law-2","name":"voyage-law-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-15","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/rerank-2.5-lite":{"id":"voyage/rerank-2.5-lite","name":"Voyage Rerank 2.5 Lite","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-3.5":{"id":"voyage/voyage-3.5","name":"voyage-3.5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-finance-2":{"id":"voyage/voyage-finance-2","name":"voyage-finance-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-06-03","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4-large":{"id":"voyage/voyage-4-large","name":"voyage-4-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/voyage-3.5-lite":{"id":"voyage/voyage-3.5-lite","name":"voyage-3.5-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4":{"id":"voyage/voyage-4","name":"voyage-4","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/voyage-3-large":{"id":"voyage/voyage-3-large","name":"voyage-3-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-07","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-code-2":{"id":"voyage/voyage-code-2","name":"voyage-code-2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-code-3":{"id":"voyage/voyage-code-3","name":"voyage-code-3","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-04","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/rerank-2.5":{"id":"voyage/rerank-2.5","name":"Voyage Rerank 2.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-4-lite":{"id":"voyage/voyage-4-lite","name":"voyage-4-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"interfaze/interfaze-beta":{"id":"interfaze/interfaze-beta","name":"Interfaze Beta","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"release_date":"2025-10-07","last_updated":"2026-04-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.5,"output":3.5}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2-fast":{"id":"zai/glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":66000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-5.3-flashx":{"id":"zai/glm-5.3-flashx","name":"GLM 5.3 FlashX","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.07,"output":0.4}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":120000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":64000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai/glm-5.3-fast":{"id":"zai/glm-5.3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"spacexai/grok-4.20-multi-agent":{"id":"spacexai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-tts":{"id":"spacexai/grok-tts","name":"Grok TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-reasoning":{"id":"spacexai/grok-4.20-reasoning","name":"Grok 4.20 Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.7":{"id":"spacexai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":1.2,"output":3.6,"cache_read":0.3,"tiers":[{"input":2.4,"output":7.2,"cache_read":0.6,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.4,"output":7.2,"cache_read":0.6}}},"spacexai/grok-voice-think-fast-1.0":{"id":"spacexai/grok-voice-think-fast-1.0","name":"Grok Voice Think Fast 1.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-non-reasoning":{"id":"spacexai/grok-4.20-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-stt":{"id":"spacexai/grok-stt","name":"Grok STT","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-imagine-image":{"id":"spacexai/grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-imagine-video":{"id":"spacexai/grok-imagine-video","name":"Grok Imagine","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.1-fast-reasoning":{"id":"spacexai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-4.3":{"id":"spacexai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-imagine-video-1.5":{"id":"spacexai/grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.5":{"id":"spacexai/grok-4.5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"spacexai/grok-voice-think-fast-2.0":{"id":"spacexai/grok-voice-think-fast-2.0","name":"Grok Voice Think Fast 2.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-imagine-image-2.0":{"id":"spacexai/grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-multi-agent-beta":{"id":"spacexai/grok-4.20-multi-agent-beta","name":"Grok 4.20 Multi Agent Beta","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.20-non-reasoning-beta":{"id":"spacexai/grok-4.20-non-reasoning-beta","name":"Grok 4.20 Beta Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.4,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-build-0.1":{"id":"spacexai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"spacexai/grok-4.1-fast-non-reasoning":{"id":"spacexai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-4.20-reasoning-beta":{"id":"spacexai/grok-4.20-reasoning-beta","name":"Grok 4.20 Beta Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"spacexai/grok-4.6":{"id":"spacexai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"prodia/flux-fast-schnell":{"id":"prodia/flux-fast-schnell","name":"Flux Schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v2":{"id":"recraft/recraft-v2","name":"Recraft V2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v4-pro":{"id":"recraft/recraft-v4-pro","name":"Recraft V4 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility":{"id":"recraft/recraft-v4.1-utility","name":"Recraft V4.1 Utility","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-pro":{"id":"recraft/recraft-v4.1-pro","name":"Recraft V4.1 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v3":{"id":"recraft/recraft-v3","name":"Recraft V3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-30","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v4.1":{"id":"recraft/recraft-v4.1","name":"Recraft V4.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-flash":{"id":"recraft/recraft-v4.1-flash","name":"Recraft V4.1 Flash","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility-pro":{"id":"recraft/recraft-v4.1-utility-pro","name":"Recraft V4.1 Utility Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4":{"id":"recraft/recraft-v4","name":"Recraft V4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"poolside/laguna-s-2.1-free":{"id":"poolside/laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude Haiku 3","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5.5-fast":{"id":"anthropic/claude-opus-5.5-fast","name":"Claude Opus 5.5 (Fast)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":8,"output":40,"cache_read":0.4,"cache_write":10}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4.8-fast":{"id":"anthropic/claude-opus-4.8-fast","name":"Claude Opus 4.8 (Fast)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-opus-5-fast":{"id":"anthropic/claude-opus-5-fast","name":"Claude Opus 5 (Fast)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"cohere/rerank-v4-fast":{"id":"cohere/rerank-v4-fast","name":"Cohere Rerank 4 Fast","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/rerank-v4-pro":{"id":"cohere/rerank-v4-pro","name":"Cohere Rerank 4 Pro","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/embed-v4.0":{"id":"cohere/embed-v4.0","name":"Embed v4.0","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":1536}},"cohere/rerank-v3.5":{"id":"cohere/rerank-v3.5","name":"Cohere Rerank 3.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.007}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.066}},"deepseek/deepseek-v3.2-thinking":{"id":"deepseek/deepseek-v3.2-thinking","name":"DeepSeek V3.2 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":128000},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"quiverai/arrow-1.1":{"id":"quiverai/arrow-1.1","name":"Arrow 1.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"quiverai/arrow-2":{"id":"quiverai/arrow-2","name":"Arrow 2","description":"Fast SVG generation model for creation, vectorization, editing, and animation","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-16","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"quiverai/arrow-2-telos":{"id":"quiverai/arrow-2-telos","name":"Arrow 2 Telos","description":"High-fidelity SVG generation model for complex vector work and long-context refinement","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-16","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"bfl/flux-kontext-max":{"id":"bfl/flux-kontext-max","name":"FLUX.1 Kontext Max","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-pro-1.1-ultra":{"id":"bfl/flux-pro-1.1-ultra","name":"FLUX1.1 [pro] Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-3-video":{"id":"bfl/flux-3-video","name":"Flux 3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-flex":{"id":"bfl/flux-2-flex","name":"FLUX.2 [flex]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-klein-4b":{"id":"bfl/flux-2-klein-4b","name":"FLUX.2 [klein] 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-pro-1.0-fill":{"id":"bfl/flux-pro-1.0-fill","name":"FLUX.1 Fill [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-2-klein-9b":{"id":"bfl/flux-2-klein-9b","name":"FLUX.2 [klein] 9B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-pro":{"id":"bfl/flux-2-pro","name":"FLUX.2 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-2-max":{"id":"bfl/flux-2-max","name":"FLUX.2 [max]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-pro-1.1":{"id":"bfl/flux-pro-1.1","name":"FLUX1.1 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-02","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-kontext-pro":{"id":"bfl/flux-kontext-pro","name":"FLUX.1 Kontext Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"tencent/hy-mt2-lite":{"id":"tencent/hy-mt2-lite","name":"Tencent Hy-MT2-Lite","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.044,"output":0.177}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"tencent/hy-mt2-plus":{"id":"tencent/hy-mt2-plus","name":"Tencent Hy-MT2-Plus","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-pro":{"id":"tencent/hy-mt2-pro","name":"Tencent Hy-MT2-Pro","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"inference-net/schematron-v2-small":{"id":"inference-net/schematron-v2-small","name":"Schematron V2 Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.23,"cache_read":0.05}},"inference-net/schematron-v2-turbo":{"id":"inference-net/schematron-v2-turbo","name":"Schematron V2 Turbo","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.15,"cache_read":0.03}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"perplexity/pplx-embed-v1-4b":{"id":"perplexity/pplx-embed-v1-4b","name":"Embed v1 4b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000}},"perplexity/pplx-embed-v1-0.6b":{"id":"perplexity/pplx-embed-v1-0.6b","name":"Embed v1 0.6b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"v0","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"muse","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/llama-3.1-8b":{"id":"meta/llama-3.1-8b","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.22,"output":0.22}},"meta/llama-3.1-70b":{"id":"meta/llama-3.1-70b","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.72,"output":0.72}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-image-1.0":{"id":"meta/muse-image-1.0","name":"Muse Image 1.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"muse","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/llama-3.3-70b":{"id":"meta/llama-3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-maverick":{"id":"meta/llama-4-maverick","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-scout":{"id":"meta/llama-4-scout","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":80000},"cost":{"input":0.25,"output":0.9}},"alibaba/qwen3.5-flash":{"id":"alibaba/qwen3.5-flash","name":"Qwen 3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"alibaba/qwen-3-32b":{"id":"alibaba/qwen-3-32b","name":"Qwen 3.32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.16,"output":0.64}},"alibaba/wan-v2.6-t2v":{"id":"alibaba/wan-v2.6-t2v","name":"Wan v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3-vl-235b-a22b-instruct":{"id":"alibaba/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen3.8-27b":{"id":"alibaba/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"alibaba/qwen3-vl-thinking":{"id":"alibaba/qwen3-vl-thinking","name":"Qwen3 VL Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen3.8-2.4t-a95b":{"id":"alibaba/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen3-next-80b-a3b-thinking":{"id":"alibaba/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.15,"output":1.2}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen3-coder-next":{"id":"alibaba/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.2}},"alibaba/qwen3-vl-instruct":{"id":"alibaba/qwen3-vl-instruct","name":"Qwen3 VL Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3-coder":{"id":"alibaba/qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.3,"tiers":[{"input":2.7,"output":13.5,"cache_read":0.54,"tier":{"type":"context","size":32001}},{"input":4.5,"output":22.5,"cache_read":0.9,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3.5-plus":{"id":"alibaba/qwen3.5-plus","name":"Qwen 3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.5,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":256001}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}}},"alibaba/qwen-3.6-max-preview":{"id":"alibaba/qwen-3.6-max-preview","name":"Qwen 3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":240000,"output":64000},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625,"tiers":[{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":128000}}]}},"alibaba/wan-v2.6-i2v-flash":{"id":"alibaba/wan-v2.6-i2v-flash","name":"Wan v2.6 Image-to-Video Flash","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v2.6-r2v-flash":{"id":"alibaba/wan-v2.6-r2v-flash","name":"Wan v2.6 Reference-to-Video Flash","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen-3-14b":{"id":"alibaba/qwen-3-14b","name":"Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.24}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen 3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"cache_write":0.125,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"cache_write":0.25,"tier":{"type":"context","size":256000}}]}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3-embedding-0.6b":{"id":"alibaba/qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen-3-30b":{"id":"alibaba/qwen-3-30b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"alibaba/qwen3.8-omni-flash":{"id":"alibaba/qwen3.8-omni-flash","name":"Qwen 3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"alibaba/qwen3-coder-30b-a3b":{"id":"alibaba/qwen3-coder-30b-a3b","name":"Qwen 3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"alibaba/qwen3-embedding-4b":{"id":"alibaba/qwen3-embedding-4b","name":"Qwen3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3-max-preview":{"id":"alibaba/qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-05","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/wan-v2.6-i2v":{"id":"alibaba/wan-v2.6-i2v","name":"Wan v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3-next-80b-a3b-instruct":{"id":"alibaba/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.15,"output":1.2}},"alibaba/wan-v2.5-t2v-preview":{"id":"alibaba/wan-v2.5-t2v-preview","name":"Wan v2.5 Text-to-Video Preview","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3-max-thinking":{"id":"alibaba/qwen3-max-thinking","name":"Qwen 3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-23","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}}]}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"tiers":[{"input":1.8,"output":9,"cache_read":0.36,"tier":{"type":"context","size":32001}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128001}},{"input":6,"output":60,"cache_read":1.2,"tier":{"type":"context","size":256001}}]}},"alibaba/qwen3.8-max-prime":{"id":"alibaba/qwen3.8-max-prime","name":"Qwen 3.8 Max Prime","description":"High-throughput edition of Qwen3.8 Max for coding, professional work, multimodal understanding, and long-running agent workflows","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":4,"output":12,"cache_read":0.5}},"alibaba/qwen3-235b-a22b-thinking":{"id":"alibaba/qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/qwen3-embedding-8b":{"id":"alibaba/qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/wan-v3.0-video":{"id":"alibaba/wan-v3.0-video","name":"Wan v3.0 Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-23","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v2.7-r2v":{"id":"alibaba/wan-v2.7-r2v","name":"Wan v2.7 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v3.0-video-prime":{"id":"alibaba/wan-v3.0-video-prime","name":"Wan v3.0 Video Prime","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.6-27b":{"id":"alibaba/qwen3.6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen-3-235b":{"id":"alibaba/qwen-3-235b","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.88}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen 3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5}}},"alibaba/wan-v2.6-r2v":{"id":"alibaba/wan-v2.6-r2v","name":"Wan v2.6 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-max-0902":{"id":"alibaba/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/wan-v2.7-t2v":{"id":"alibaba/wan-v2.7-t2v","name":"Wan v2.7 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"mixedbread/toast-1":{"id":"mixedbread/toast-1","name":"Toast 1","description":"Specialized search model for knowledge-intensive questions, multi-step retrieval, and evidence synthesis","attachment":false,"reasoning":false,"tool_call":true,"release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":4000},"cost":{"input":0.3,"output":0.72,"cache_read":0.036}},"klingai/kling-v2.6-t2v":{"id":"klingai/kling-v2.6-t2v","name":"Kling v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-i2v":{"id":"klingai/kling-v2.6-i2v","name":"Kling v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.5-turbo-t2v":{"id":"klingai/kling-v2.5-turbo-t2v","name":"Kling v2.5 Turbo Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-i2v":{"id":"klingai/kling-v3.0-i2v","name":"Kling v3.0 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-motion-control":{"id":"klingai/kling-v3.0-motion-control","name":"Kling v3.0 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-t2v":{"id":"klingai/kling-v3.0-t2v","name":"Kling v3.0 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-motion-control":{"id":"klingai/kling-v2.6-motion-control","name":"Kling v2.6 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.5-turbo-i2v":{"id":"klingai/kling-v2.5-turbo-i2v","name":"Kling v2.5 Turbo Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-lite":{"id":"bytedance/seedream-5.0-lite","name":"Seedream 5.0 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-4.5":{"id":"bytedance/seedream-4.5","name":"Seedream 4.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-11-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seed-1.8":{"id":"bytedance/seed-1.8","name":"Seed 1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"cost":{"input":0.25,"output":2,"cache_read":0.05,"tiers":[{"input":0.5,"output":4,"cache_read":0.05,"tier":{"type":"context","size":128001}}]}},"bytedance/seed-2.1-turbo":{"id":"bytedance/seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.5,"cache_read":0.1}},"bytedance/seedream-4.0":{"id":"bytedance/seedream-4.0","name":"Seedream 4.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro":{"id":"bytedance/seedance-v1.0-pro","name":"Seedance v1.0 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-11","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.0":{"id":"bytedance/seedance-2.0","name":"Seedance 2.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.0-fast":{"id":"bytedance/seedance-2.0-fast","name":"Seedance 2.0 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.5-pro":{"id":"bytedance/seedance-v1.5-pro","name":"Seedance v1.5 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.0-mini":{"id":"bytedance/seedance-2.0-mini","name":"Seedance 2.0 Mini","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.5":{"id":"bytedance/seedance-2.5","name":"Seedance 2.5","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro-fast":{"id":"bytedance/seedance-v1.0-pro-fast","name":"Seedance v1.0 Pro Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-31","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-pro":{"id":"bytedance/seedream-5.0-pro","name":"Seedream 5.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-11","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seed-1.6":{"id":"bytedance/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.25,"output":2,"cache_read":0.05,"tiers":[{"input":0.5,"output":4,"cache_read":0.05,"tier":{"type":"context","size":128001}}]}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemini-omni-flash-preview":{"id":"google/gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":57920},"cost":{"input":1.5,"output":9}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Gemini 3.1 Flash Image Preview (Nano Banana 2)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.5-transcribe-live":{"id":"google/gemini-3.5-transcribe-live","name":"Gemini 3.5 Transcribe Live","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana (Gemini 2.5 Flash Image)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.8-flash-lite-tts":{"id":"google/gemini-3.8-flash-lite-tts","name":"Gemini 3.8 Flash-Lite TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.5,"output":6,"cache_read":0.125}},"google/text-multilingual-embedding-002":{"id":"google/text-multilingual-embedding-002","name":"Text Multilingual Embedding 002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-01","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-embedding-2":{"id":"google/gemini-embedding-2","name":"Gemini Embedding 2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/veo-3.1-lite-generate-001":{"id":"google/veo-3.1-lite-generate-001","name":"Veo 3.1 Lite Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200001}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/veo-3.1-fast-generate-001":{"id":"google/veo-3.1-fast-generate-001","name":"Veo 3.1 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/veo-3.0-fast-generate-001":{"id":"google/veo-3.0-fast-generate-001","name":"Veo 3.0 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.8-live":{"id":"google/gemini-3.8-live","name":"Gemini 3.8 Live","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-15","last_updated":"2026-09-15","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.75,"output":4.5}},"google/veo-3.1-generate-001":{"id":"google/veo-3.1-generate-001","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/text-embedding-005":{"id":"google/text-embedding-005","name":"Text Embedding 005","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-01","last_updated":"2024-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.5-transcribe":{"id":"google/gemini-3.5-transcribe","name":"Gemini 3.5 Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":12}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.8-flash-tts":{"id":"google/gemini-3.8-flash-tts","name":"Gemini 3.8 Flash TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.5,"output":9,"cache_read":0.125}},"google/gemini-3.8-live-extended-thinking":{"id":"google/gemini-3.8-live-extended-thinking","name":"Gemini 3.8 Live Extended Thinking","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-15","last_updated":"2026-09-15","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.75,"output":4.5}},"google/veo-3.0-generate-001":{"id":"google/veo-3.0-generate-001","name":"Veo 3.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"sakana/fugu-max":{"id":"sakana/fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"sakana/fugu-ultra-v2":{"id":"sakana/fugu-ultra-v2","name":"Fugu Ultra v2","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/namazu":{"id":"sakana/namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"inclusionai/ling-3.0-flash-fin-free":{"id":"inclusionai/ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin (Free)","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.021,"output":0.063,"cache_read":0.0042}},"inclusionai/ling-3.0-flash-sante":{"id":"inclusionai/ling-3.0-flash-sante","name":"Ling 3.0 Flash Sante","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante-free":{"id":"inclusionai/ling-3.0-flash-sante-free","name":"Ling 3.0 Flash Sante (Free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k3-fast":{"id":"moonshotai/kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":216144,"output":216144},"cost":{"input":0.47,"output":2,"cache_read":0.141}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nvidia Nemotron Nano 9B V2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.06,"output":0.23}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.15,"output":0.65}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nvidia Nemotron Nano 12B V2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.6}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo V2.6 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo M2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131100},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.6-pro-ultraspeed":{"id":"xiaomi/mimo-v2.6-pro-ultraspeed","name":"MiMo V2.6 Pro UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo V2.6 Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-23","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-h3":{"id":"minimax/minimax-h3","name":"MiniMax H3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"Minimax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-h3-max":{"id":"minimax/minimax-h3-max","name":"MiniMax H3 Max","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 High Speed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 High Speed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":196608},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"stepfun/step-5-preview":{"id":"stepfun/step-5-preview","name":"Step 5 Preview","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"StepFun 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"inception/mercury-coder-small":{"id":"inception/mercury-coder-small","name":"Mercury Coder Small Beta","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"mercury","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-02-26","last_updated":"2025-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":16384},"cost":{"input":0.25,"output":1}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.024999999999999998}},"amazon/titan-embed-text-v2":{"id":"amazon/titan-embed-text-v2","name":"Titan Text Embeddings V2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"titan-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-30","last_updated":"2024-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"amazon/nova-2-lite":{"id":"amazon/nova-2-lite","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2024-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}},"amazon/nova-lite":{"id":"amazon/nova-lite","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"amazon/nova-pro":{"id":"amazon/nova-pro","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"amazon/nova-micro":{"id":"amazon/nova-micro","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"mistral/codestral-embed":{"id":"mistral/codestral-embed","name":"Codestral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"codestral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/mistral-nemo":{"id":"mistral/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-07-18","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":60288,"output":16000},"cost":{"input":0.04,"output":0.17}},"mistral/mistral-medium-3.5":{"id":"mistral/mistral-medium-3.5","name":"Mistral Medium Latest","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-05-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-small":{"id":"mistral/mistral-small","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2024-09-17","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-large-3":{"id":"mistral/mistral-large-3","name":"Mistral Large 3","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/mistral-embed":{"id":"mistral/mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/ministral-14b":{"id":"mistral/ministral-14b","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistral/codestral":{"id":"mistral/codestral","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/ministral-8b":{"id":"mistral/ministral-8b","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"mistral/ministral-3b":{"id":"mistral/ministral-3b","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"fish-audio/s2.1-pro":{"id":"fish-audio/s2.1-pro","name":"S2.1 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-28","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s1":{"id":"fish-audio/s1","name":"S1","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/transcribe-1":{"id":"fish-audio/transcribe-1","name":"Transcribe-1","description":"Speech transcription model for accurate audio-to-text and captioning workflows","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2-pro":{"id":"fish-audio/s2-pro","name":"S2 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"openai/gpt-4.1-mini-fast":{"id":"openai/gpt-4.1-mini-fast","name":"GPT-4.1 mini (Fast)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.7,"output":2.8,"cache_read":0.175}},"openai/o3-fast":{"id":"openai/o3-fast","name":"o3 (Fast)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT 5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.15,"output":0.6}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":12289,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.2-fast":{"id":"openai/gpt-5.2-fast","name":"GPT 5.2 (Fast)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2.5,"output":10}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT 5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/text-embedding-3-small":{"id":"openai/text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-realtime-2.1":{"id":"openai/gpt-realtime-2.1","name":"gpt-realtime-2.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-fast":{"id":"openai/gpt-4o-fast","name":"GPT-4o (Fast)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":4.25,"output":17,"cache_read":2.125}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-live-1":{"id":"openai/gpt-live-1","name":"GPT-Live 1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT-Realtime-1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":16,"cache_read":0.4}},"openai/gpt-4.1-nano-fast":{"id":"openai/gpt-4.1-nano-fast","name":"GPT-4.1 nano (Fast)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-5.6-luna-fast":{"id":"openai/gpt-5.6-luna-fast","name":"GPT 5.6 Luna (Fast)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":0.8,"output":3.6,"cache_read":0.08,"cache_write":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.8,"output":3.6,"cache_read":0.08,"cache_write":1}}},"openai/gpt-6-astra-fast":{"id":"openai/gpt-6-astra-fast","name":"GPT-6 Astra (Fast)","description":"Fast variant of GPT-6 Astra for low-latency assistance and high-volume workloads.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25,"tiers":[{"input":40,"output":150,"cache_read":4,"cache_write":50,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":40,"output":150,"cache_read":4,"cache_write":50}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT 5.2 ","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-5.1-thinking-fast":{"id":"openai/gpt-5.1-thinking-fast","name":"GPT 5.1 Thinking (Fast)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/text-embedding-ada-002":{"id":"openai/text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/whisper-1":{"id":"openai/whisper-1","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-09-21","last_updated":"2022-09-21","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.3-codex-fast":{"id":"openai/gpt-5.3-codex-fast","name":"GPT 5.3 Codex (Fast)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/gpt-5.5-fast":{"id":"openai/gpt-5.5-fast","name":"GPT 5.5 (Fast)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":12.5,"output":75,"cache_read":1.25}},"openai/gpt-image-2.5-flare":{"id":"openai/gpt-image-2.5-flare","name":"GPT Image 2.5 Flare","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-5.4-fast":{"id":"openai/gpt-5.4-fast","name":"GPT 5.4 (Fast)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.6-sol-fast":{"id":"openai/gpt-5.6-sol-fast","name":"GPT 5.6 Sol (Fast)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10,"tiers":[{"input":16,"output":60,"cache_read":1.6,"cache_write":20,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":16,"output":60,"cache_read":1.6,"cache_write":20}}},"openai/o4-mini-fast":{"id":"openai/o4-mini-fast","name":"o4-mini (Fast)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":128000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1-thinking":{"id":"openai/gpt-5.1-thinking","name":"GPT 5.1 Thinking","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-11-12","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/text-embedding-3-large":{"id":"openai/text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.4-mini-fast":{"id":"openai/gpt-5.4-mini-fast","name":"GPT 5.4 Mini (Fast)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"openai/gpt-realtime-mini":{"id":"openai/gpt-realtime-mini","name":"GPT-Realtime mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"openai/gpt-5-mini-fast":{"id":"openai/gpt-5-mini-fast","name":"GPT-5 mini (Fast)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.45,"output":3.6,"cache_read":0.045}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3 Pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":122880,"output":8192},"cost":{"input":0.03,"output":0.14}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.07,"output":0.2}},"openai/gpt-image-2.5-sunburst":{"id":"openai/gpt-image-2.5-sunburst","name":"GPT Image 2.5 Sunburst","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-5.6-terra-fast":{"id":"openai/gpt-5.6-terra-fast","name":"GPT 5.6 Terra (Fast)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":36,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":36,"cache_read":0.8,"cache_write":10}}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT Image 1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-6-sol-fast":{"id":"openai/gpt-6-sol-fast","name":"GPT-6 Sol (Fast)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/tts-1":{"id":"openai/tts-1","name":"TTS-1","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-6-luna-fast":{"id":"openai/gpt-6-luna-fast","name":"GPT-6 Luna (Fast)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.5,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":0.4,"output":1.5,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5-fast":{"id":"openai/gpt-5-fast","name":"GPT-5 (Fast)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/gpt-realtime-2":{"id":"openai/gpt-realtime-2","name":"gpt-realtime-2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/gpt-4.1-fast":{"id":"openai/gpt-4.1-fast","name":"GPT-4.1 (Fast)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT Image 1 Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":8,"cache_read":0.2}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":1.25,"output":5}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5,"cache_read":0.1}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-4o-mini-fast":{"id":"openai/gpt-4o-mini-fast","name":"GPT-4o mini (Fast)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"openai/gpt-realtime-whisper":{"id":"openai/gpt-realtime-whisper","name":"gpt-realtime-whisper","description":"Streaming speech-to-text model for low-latency transcript deltas from live audio","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/tts-1-hd":{"id":"openai/tts-1-hd","name":"TTS-1 HD","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}}}},"zai-coding-plan":{"id":"zai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/coding/paas/v4","name":"Z.AI Coding Plan","doc":"https://docs.z.ai/devpack/overview","models":{"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2-highspeed":{"id":"glm-5.2-highspeed","name":"GLM-5.2 Highspeed","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"ebcloud":{"id":"ebcloud","env":["EBCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://maas-api.ebcloud.com/v1","name":"EBCloud","doc":"https://docs.ebtech.com/ai/model-api.html","models":{"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9286,"output":3.8571}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.8571,"output":3.4286}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.143,"output":0.2857}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4286,"output":0.8571}}}},"greenpt":{"id":"greenpt","env":["GREENPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.greenpt.ai/v1","name":"GreenPT","doc":"https://docs.greenpt.ai","models":{"glm-5.2-honey-ultra":{"id":"glm-5.2-honey-ultra","name":"GLM-5.2 Honey Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-ponytail-lite":{"id":"glm-5.2-ponytail-lite","name":"GLM-5.2 Ponytail Lite","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"Gemma 3 27B","description":"Google Gemma 3 multimodal model for chat, reasoning, and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40000,"output":8192},"cost":{"input":0.342,"output":0.684}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.127754,"output":0.511016,"cache_read":0.0255508}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1596,"output":0.399,"cache_read":0.0456}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen3 235B MoE instruct model for long-context multilingual chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1.026,"output":3.078}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.762,"output":18.81,"cache_read":0.9405}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.255552,"output":1.27776,"cache_read":0.0127776}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.1938,"output":1.129,"cache_read":0.0627}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7524,"output":4.275,"cache_read":0.2508}},"green-s":{"id":"green-s","name":"Green S","description":"GreenPT speech-to-text model for pre-recorded and live transcription","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"green-l":{"id":"green-l","name":"Green L","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":1.083}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":1.254,"output":1.254}},"gemma4":{"id":"gemma4","name":"gemma4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.57,"output":1.71}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.342,"output":2.052}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.285,"output":0.285}},"glm-5.2-honey":{"id":"glm-5.2-honey","name":"GLM-5.2 Honey","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"holo2-30b-a3b":{"id":"holo2-30b-a3b","name":"Holo2 30B A3B","description":"H Company Holo2 vision model for GUI navigation and computer-use agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11","last_updated":"2025-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":22016,"output":16384},"cost":{"input":0.399,"output":0.969}},"green-r-raw":{"id":"green-r-raw","name":"Green R Raw","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"green-s-pro":{"id":"green-s-pro","name":"Green S Pro","description":"GreenPT advanced speech-to-text model with multilingual transcription support","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-02","last_updated":"2025-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"devstral-2-123b-instruct-2512":{"id":"devstral-2-123b-instruct-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":16384},"cost":{"input":0.57,"output":2.736}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":2.052,"output":10.26}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.798,"output":4.959}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"voxtral-small-24b-2507":{"id":"voxtral-small-24b-2507","name":"Voxtral Small 24B","description":"Mistral Voxtral audio-understanding model for speech and transcription tasks","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.228,"output":0.513}},"green-l-raw":{"id":"green-l-raw","name":"Green L Raw","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"glm-5.2-ponytail-ultra":{"id":"glm-5.2-ponytail-ultra","name":"GLM-5.2 Ponytail Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"green-r":{"id":"green-r","name":"Green R","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"status":"deprecated","cost":{"input":1.756,"output":5.518}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.228,"output":0.456}},"glm-5.2-caveman":{"id":"glm-5.2-caveman","name":"GLM-5.2 Caveman","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k2.6-fast":{"id":"kimi-k2.6-fast","name":"Kimi K2.6 Fast","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":1.655,"output":8.778}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.228,"output":0.798}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.27754,"output":5.11016,"cache_read":0.319385}},"glm-5.2-caveman-ultra":{"id":"glm-5.2-caveman-ultra","name":"GLM-5.2 Caveman Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9006,"output":4.389,"cache_read":0.1881}},"glm-5.2-ponytail":{"id":"glm-5.2-ponytail","name":"GLM-5.2 Ponytail","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-honey-lite":{"id":"glm-5.2-honey-lite","name":"GLM-5.2 Honey Lite","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-caveman-lite":{"id":"glm-5.2-caveman-lite","name":"GLM-5.2 Caveman Lite","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}}}},"mixlayer":{"id":"mixlayer","env":["MIXLAYER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.mixlayer.ai/v1","name":"Mixlayer","doc":"https://docs.mixlayer.com","models":{"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3.2}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.3}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6}}}},"hyper":{"id":"hyper","env":["HYPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://hyper.charm.land/v1","name":"Charm Hyper","doc":"https://hyper.charm.land","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16332,"output":0.5444,"cache_read":0.031575}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.044}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16000},"cost":{"input":3.2664,"output":16.332,"cache_read":0.32664}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":64000},"cost":{"input":0.2,"output":0.8,"cache_read":0.04}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":26214},"cost":{"input":0.6,"output":2.5,"cache_read":0.3}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":25600},"cost":{"input":0.098,"output":0.334,"cache_read":0.049}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":6553},"cost":{"input":0.484,"output":1.852,"cache_read":0.242}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":1.437216,"output":4.311648,"cache_read":0.047907}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-15","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1.0888,"output":4.40964,"cache_read":0.185096}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.32664,"output":1.30656,"cache_read":0.064239}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.152432}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-13","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":13107},"cost":{"input":0.178,"output":0.68,"cache_read":0.089}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.283088}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-07-03","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16000},"cost":{"input":1.03436,"output":4.3552,"cache_read":0.206872}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-15","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.2,"output":4.8,"cache_read":0.24}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}}}},"jalapeno":{"id":"jalapeno","env":["JALAPENO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jalapeno-cloud.ai/v1","name":"Jalapeno Cloud","doc":"https://www.jalapeno-cloud.ai/docs/","models":{"Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.3,"output":1.5}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.15,"output":1.5}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.38,"output":4.4}},"Hy3":{"id":"Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen3.5-27B":{"id":"Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.6,"output":3.38}},"Qwen3.5-35B-A3B":{"id":"Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"Qwen3.5-122B-A10B":{"id":"Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen3.5-397B-A17B":{"id":"Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":271360,"output":262144},"cost":{"input":0.95,"output":4}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":180224},"cost":{"input":0.6,"output":3}}}},"dinference":{"id":"dinference","env":["DINFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.dinference.com/v1","name":"DInference","doc":"https://dinference.com","models":{"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.75,"output":2.4}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.22,"output":0.88}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.45,"output":1.65}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":3.89}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.25,"output":3.89}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08","last_updated":"2025-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.0675,"output":0.27}}}},"neosmith":{"id":"neosmith","env":["NEOSMITH_API_KEY"],"npm":"@ai-sdk/openai","api":"https://router.neosmith.ai/v1","name":"NeoSmith","doc":"https://neosmith.ai/docs","models":{"neosmith.neolite":{"id":"neosmith.neolite","name":"NeoSmith NeoLite","description":"Sealed single-model budget tier. 512K context, text and images, tool use, and no escalation of any kind.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":64000},"cost":{"input":0.6,"output":2.4,"cache_read":0.08,"cache_write":0}},"neosmith.intelligent-maestro":{"id":"neosmith.intelligent-maestro","name":"NeoSmith Maestro","description":"Highest-accuracy coding tier. Hard, self-contained problems run NeoSmith's premium multi-model solver; everything else gets the strongest intelligence tier.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.4,"output":12,"cache_read":0.35,"cache_write":0}},"neosmith.intelligent-pro":{"id":"neosmith.intelligent-pro","name":"NeoSmith Pro","description":"Default production tier. Intelligent NeoSmith routing with a Claude Opus ceiling on escalation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.81,"output":8.39,"cache_read":0.3,"cache_write":0}},"neosmith.intelligent-basic":{"id":"neosmith.intelligent-basic","name":"NeoSmith Basic","description":"Cost-capped tier. Intelligent routing with a Claude Sonnet ceiling — Opus is never invoked.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.17,"output":4.37,"cache_read":0.22,"cache_write":0}}}},"fireworks-ai":{"id":"fireworks-ai","env":["FIREWORKS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.fireworks.ai/inference/v1/","name":"Fireworks AI","doc":"https://fireworks.ai/docs/","models":{"accounts/fireworks/models/kimi-k2p6":{"id":"accounts/fireworks/models/kimi-k2p6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.5,"output":6,"cache_read":0.22},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"accounts/fireworks/models/nemotron-3-ultra-nvfp4":{"id":"accounts/fireworks/models/nemotron-3-ultra-nvfp4","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/muse-glimmer-30b":{"id":"accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"accounts/fireworks/models/kimi-k3":{"id":"accounts/fireworks/models/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3.75,"output":18.75,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/models/qwen3p8-2p4t-a95b":{"id":"accounts/fireworks/models/qwen3p8-2p4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/models/ember-1":{"id":"accounts/fireworks/models/ember-1","name":"Ember-1","description":"Specialized model from Fireworks built on Kimi K3, producing shorter reasoning traces with approximately 40% fewer tokens while maintaining comparable quality","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/models/deepseek-v4-flash-vision-exp":{"id":"accounts/fireworks/models/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/deepseek-v4p1-flash":{"id":"accounts/fireworks/models/deepseek-v4p1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/qwen3p7-plus":{"id":"accounts/fireworks/models/qwen3p7-plus","name":"Qwen 3.7 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08}},"accounts/fireworks/models/glm-5p3":{"id":"accounts/fireworks/models/glm-5p3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":262144},"experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.325},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"accounts/fireworks/models/minimax-m2p7":{"id":"accounts/fireworks/models/minimax-m2p7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","provider":{"body":{"service_tier":"priority"}},"cost":{"input":1.2,"output":1.2,"cache_read":0.6}},"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b":{"id":"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.65,"output":4.95,"cache_read":0.055},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"accounts/fireworks/models/inkling":{"id":"accounts/fireworks/models/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"accounts/fireworks/models/glm-5p2":{"id":"accounts/fireworks/models/glm-5p2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.175},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"accounts/fireworks/models/minimax-m3":{"id":"accounts/fireworks/models/minimax-m3","name":"MiniMax-M3","description":"Fireworks text-only MiniMax coding model for long-context reasoning and agent tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"experimental":{"modes":{"priority":{"cost":{"input":0.45,"output":1.8,"cache_read":0.09},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"accounts/fireworks/models/kimi-k2p7-code":{"id":"accounts/fireworks/models/kimi-k2p7-code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.425,"output":6,"cache_read":0.285},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"accounts/fireworks/models/glm-5p3-flash":{"id":"accounts/fireworks/models/glm-5p3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":0.1875,"output":0.625,"cache_read":0.0375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"accounts/fireworks/models/deepseek-v4-pro":{"id":"accounts/fireworks/models/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"deprecated","experimental":{"modes":{"priority":{"cost":{"input":1.2,"output":1.2,"cache_read":0.6},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.2,"output":1.2,"cache_read":0.6}},"accounts/fireworks/models/gpt-oss-120b":{"id":"accounts/fireworks/models/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"experimental":{"modes":{"priority":{"cost":{"input":0.18,"output":0.72,"cache_read":0.018},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"accounts/fireworks/models/qwen3p8-max":{"id":"accounts/fireworks/models/qwen3p8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3,"output":9,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/routers/minimax-latest":{"id":"accounts/fireworks/routers/minimax-latest","name":"MiniMax Latest","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"experimental":{"modes":{"priority":{"cost":{"input":0.45,"output":1.8,"cache_read":0.09},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"accounts/fireworks/routers/glm-flash-latest":{"id":"accounts/fireworks/routers/glm-flash-latest","name":"GLM Flash Latest (GLM 5.3 Flash)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":0.1875,"output":0.625,"cache_read":0.0375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"accounts/fireworks/routers/glm-fast-latest":{"id":"accounts/fireworks/routers/glm-fast-latest","name":"GLM 5.3 Fast (Latest)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048572,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.39}},"accounts/fireworks/routers/kimi-k3-fast":{"id":"accounts/fireworks/routers/kimi-k3-fast","name":"Kimi K3 Fast","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"accounts/fireworks/routers/glm-5p3-fast":{"id":"accounts/fireworks/routers/glm-5p3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048572,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.39}},"accounts/fireworks/routers/kimi-fast-latest":{"id":"accounts/fireworks/routers/kimi-fast-latest","name":"Kimi Fast Latest","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"accounts/fireworks/routers/qwen-max-latest":{"id":"accounts/fireworks/routers/qwen-max-latest","name":"Qwen Max Latest (Qwen3.8 Max)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3,"output":9,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/routers/kimi-latest":{"id":"accounts/fireworks/routers/kimi-latest","name":"Kimi Latest","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"experimental":{"modes":{"priority":{"cost":{"input":3.75,"output":18.75,"cache_read":0.375},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/routers/deepseek-flash-latest":{"id":"accounts/fireworks/routers/deepseek-flash-latest","name":"DeepSeek Flash Latest","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":0.275,"output":0.825,"cache_read":0.00875},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/routers/glm-5p2-fast":{"id":"accounts/fireworks/routers/glm-5p2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"accounts/fireworks/routers/deepseek-pro-latest":{"id":"accounts/fireworks/routers/deepseek-pro-latest","name":"DeepSeek Pro Latest","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"experimental":{"modes":{"priority":{"cost":{"input":1.65,"output":4.95,"cache_read":0.055},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"accounts/fireworks/routers/glm-latest":{"id":"accounts/fireworks/routers/glm-latest","name":"GLM Latest","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":262144},"experimental":{"modes":{"priority":{"cost":{"input":1.75,"output":5.5,"cache_read":0.325},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"stepfun-ai":{"id":"stepfun-ai","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/v1","name":"StepFun (Global)","doc":"https://platform.stepfun.ai/docs/en/overview/concept","models":{"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"input":1024000,"output":65536},"cost":{"input":1,"output":2.7,"cache_read":0.05}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}}}},"fastrouter":{"id":"fastrouter","env":["FASTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://go.fastrouter.ai/api/v1","name":"FastRouter","doc":"https://fastrouter.ai/models","models":{"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.95,"output":3.15}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.05,"output":3.5}},"sarvam/sarvam-30b":{"id":"sarvam/sarvam-30b","name":"Sarvam 30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.1}},"sarvam/sarvam-105b":{"id":"sarvam/sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"x-ai/grok-4":{"id":"x-ai/grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.75,"cache_write":15}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2}},"bytedance/seedance-2":{"id":"bytedance/seedance-2","name":"Seedance 2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":4096,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"google/veo3.1":{"id":"google/veo3.1","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/imagen-4.0-ultra":{"id":"google/imagen-4.0-ultra","name":"Imagen 4 Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo3.1-lite":{"id":"google/veo3.1-lite","name":"Veo 3.1 Lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"google/veo3.1-fast":{"id":"google/veo3.1-fast","name":"Veo 3.1 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.0375}},"google/imagen-4.0-fast":{"id":"google/imagen-4.0-fast","name":"Imagen 4 Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"deepseek-ai/deepseek-r1-distill-llama-70b":{"id":"deepseek-ai/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.14}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.55,"output":2.2}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4}},"wanx/wan-v2-6":{"id":"wanx/wan-v2-6","name":"Wan 2.6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":true,"limit":{"context":400000,"output":0}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1.2}},"leonardo-ai/lucid-realism":{"id":"leonardo-ai/lucid-realism","name":"Lucid Realism","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"leonardo-ai/lucid-origin":{"id":"leonardo-ai/lucid-origin","name":"Lucid Origin","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT Realtime 1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32000,"output":4096},"cost":{"input":4,"output":16}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.05,"output":0.2}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}}}},"orcarouter":{"id":"orcarouter","env":["ORCAROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.orcarouter.ai/v1","name":"OrcaRouter","doc":"https://docs.orcarouter.ai","models":{"grok/grok-4.3":{"id":"grok/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok/grok-4.5":{"id":"grok/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"grok/grok-4.6":{"id":"grok/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":10}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"deepseek/deepseek-v4-flash-free":{"id":"deepseek/deepseek-v4-flash-free","name":"DeepSeek V4 Flash (free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek/deepseek-reasoner":{"id":"deepseek/deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"tencent/hy3-free":{"id":"tencent/hy3-free","name":"Hy3 (free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.18,"output":0.59,"cache_read":0.059}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.075,"output":0.25}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.26,"cache_write":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5.3-flash-free":{"id":"z-ai/glm-5.3-flash-free","name":"GLM-5.3-Flash (free)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38,"cache_read":0.02}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-robotics-er-1.6-preview":{"id":"google/gemini-robotics-er-1.6-preview","name":"Gemini Robotics-ER 1.6 Preview","description":"Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.33,"cache_read":0.0075}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333,"input_audio":3}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"kimi/kimi-k3":{"id":"kimi/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.3,"output":16.5,"cache_read":0.33}},"kimi/kimi-k2.6":{"id":"kimi/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1,"cache_write":0}},"kimi/kimi-k2.7-code":{"id":"kimi/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"orcarouter/fusion-mini":{"id":"orcarouter/fusion-mini","name":"OrcaRouter Fusion Mini","description":"Leaner two-model Fusion panel that runs Claude Opus 4.8 and GPT-5.5 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/free":{"id":"orcarouter/free","name":"OrcaRouter Free","description":"Built-in router over the free tier that scores each request's difficulty and sends light work to the smaller free model and harder work to the stronger one. Priced at zero and never falls back to a paid model.","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0,"output":0}},"orcarouter/fusion":{"id":"orcarouter/fusion","name":"OrcaRouter Fusion","description":"Curated fan-out router that runs Claude Opus 4.8, GPT-5.5 and Gemini 3.1 Pro in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/fusion-flash":{"id":"orcarouter/fusion-flash","name":"OrcaRouter Fusion Flash","description":"Budget Fusion panel that runs Gemini 3.5 Flash, MiniMax M2.7 and GLM 5.1 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Cost-sensitive fan-out over a 200K window.","family":"model-router","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"orcarouter/auto":{"id":"orcarouter/auto","name":"OrcaRouter Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2026-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.563}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.33,"output":2.4}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.4,"output":4}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.086,"output":0.688}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.115,"output":0.688,"reasoning":2.4}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.115,"output":0.917}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.057,"output":0.459}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.35,"output":1.42,"cache_read":0.071}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":100000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":100000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.17}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}}}},"friendli":{"id":"friendli","env":["FRIENDLI_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.friendli.ai/serverless/v1","name":"Friendli","doc":"https://friendli.ai/docs/guides/serverless_endpoints/introduction","models":{"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}}}},"kimi-code-plan-cn":{"id":"kimi-code-plan-cn","env":["KIMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kimi.com/coding/v1","name":"Kimi For Coding (kimi.com)","doc":"https://www.kimi.com/code/docs/en/kimi-code/models.html","models":{"kimi-for-coding-highspeed":{"id":"kimi-for-coding-highspeed","name":"Kimi For Coding HighSpeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-for-coding":{"id":"kimi-for-coding","name":"kimi-for-coding","description":"Kimi coding model with more efficient thinking and up to 1M context, available through Kimi Code","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3-256k":{"id":"k3-256k","name":"Kimi K3-256K","description":"256K-context version of Kimi K3, reducing token consumption for shorter coding sessions","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3":{"id":"k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"inco":{"id":"inco","env":["INCO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inco.ai/v1","name":"Inco","doc":"https://platform.inco.ai/docs","models":{"kimi-k3:fast":{"id":"kimi-k3:fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":6,"output":30}},"glm-5.3:fast":{"id":"glm-5.3:fast","name":"GLM-5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.8,"output":8.8}},"glm-5.3-flash:fast":{"id":"glm-5.3-flash:fast","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2}},"deepseek-v4.1-flash:fast":{"id":"deepseek-v4.1-flash:fast","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.6,"output":2.4}},"minimax-m3:fast":{"id":"minimax-m3:fast","name":"MiniMax M3 Fast","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}}}},"sakana":{"id":"sakana","env":["SAKANA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sakana.ai/v1","name":"Sakana AI","doc":"https://console.sakana.ai/models","models":{"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"fugu-ultra-20260615":{"id":"fugu-ultra-20260615","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana-namazu":{"id":"sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"fugu":{"id":"fugu","name":"Fugu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"}}}},"scx-ai":{"id":"scx-ai","env":["SCX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scx.ai/v1","name":"SCX.ai","doc":"https://platform.scx.ai/docs","models":{"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.55,"output":1.784,"cache_read":0.111}},"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":983616,"output":131072},"cost":{"input":1.815,"output":5.4461,"cache_read":0.17,"cache_write":2.5}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.17,"output":0.55}}}},"zenifra":{"id":"zenifra","env":["ZENIFRA_AI_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai.zenifra.com/v1","name":"Zenifra","doc":"https://docs.zenifra.com","models":{"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"provider":{"shape":"completions"},"cost":{"input":0.19,"output":0.48}}}},"tokenrouter":{"id":"tokenrouter","env":["TOKENROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenrouter.com/v1","name":"TokenRouter","doc":"https://www.tokenrouter.com/docs/tokenrouter-feature-guide/","models":{"z-ai/glm-5.3-free":{"id":"z-ai/glm-5.3-free","name":"GLM-5.3 (free)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"google-vertex-anthropic":{"id":"google-vertex-anthropic","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex/anthropic","name":"Vertex (Anthropic)","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude","models":{"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-opus-5-5@default":{"id":"claude-opus-5-5@default","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}}}},"moonshotai":{"id":"moonshotai","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.ai/v1","name":"Moonshot AI","doc":"https://platform.moonshot.ai/docs/api/chat","models":{"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"ainetcafe":{"id":"ainetcafe","env":["AINETCAFE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://microquickjs.com/v1","name":"ainetcafe","doc":"https://ainetcafe.com/k3/guides/","models":{"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":2.1,"output":10.5,"cache_read":0.3}}}},"wallaby":{"id":"wallaby","env":["WALLABY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.wallabytoken.com/v1","name":"Wallaby","doc":"https://wallabytoken.com/docs","models":{"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.7,"output":13.5,"cache_read":0.27}}}},"scnet-token-plan":{"id":"scnet-token-plan","env":["SCNET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scnet.cn/api/llm/v1","name":"SCNet Token Plan","doc":"https://www.scnet.cn/ac/openapi/doc/2.0/moduleapi/plans/token-plan.html","models":{"DeepSeek-V4-Flash-0731":{"id":"DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro-0813":{"id":"DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4.1-Flash":{"id":"DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5":{"id":"GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3-Flash":{"id":"GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Flash":{"id":"Qwen3.8-Flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3":{"id":"GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}}}},"ofox":{"id":"ofox","env":["OFOX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ofox.ai/v1","name":"Ofox","doc":"https://ofox.ai/docs","models":{"bailian/qwen-flash":{"id":"bailian/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.22,"cache_read":0.0043,"cache_write":0.027}},"bailian/qwen3.5-flash":{"id":"bailian/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"bailian/qwen3.7-max":{"id":"bailian/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"bailian/qwen3.8-27b":{"id":"bailian/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.05,"cache_write":0.5625}},"bailian/qwen3.5-27b":{"id":"bailian/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"bailian/qwen-max":{"id":"bailian/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.35,"output":1.38,"cache_read":0.069}},"bailian/qwen3.8-max":{"id":"bailian/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"bailian/qwen3-coder-next":{"id":"bailian/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"bailian/qwen3.5-plus":{"id":"bailian/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"bailian/qwen-plus":{"id":"bailian/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"bailian/qwen3-max":{"id":"bailian/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.36,"output":1.43,"cache_read":0.072}},"bailian/qwen-turbo":{"id":"bailian/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.05,"output":0.09,"cache_read":0.0086}},"bailian/qwen-vl-max":{"id":"bailian/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.58,"cache_read":0.046}},"bailian/qwen3.5-122b-a10b":{"id":"bailian/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.29,"cache_read":0.29}},"bailian/qwen3.6-flash":{"id":"bailian/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"bailian/qwen3-coder-flash":{"id":"bailian/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.06,"cache_write":0.27}},"bailian/qwen3-coder-plus":{"id":"bailian/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":9,"cache_read":0.2,"cache_write":1}},"bailian/qwen3.8-flash":{"id":"bailian/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"bailian/qwen3.5-35b-a3b":{"id":"bailian/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"bailian/qwen3.6-max-preview":{"id":"bailian/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"bailian/qwen3.5-397b-a17b":{"id":"bailian/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"bailian/qwen3.6-27b":{"id":"bailian/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.6,"output":3.6}},"bailian/qwen3.6-plus":{"id":"bailian/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"bailian/qwen3.7-plus":{"id":"bailian/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"bailian/qwen3.8-max-0902":{"id":"bailian/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5.5":{"id":"anthropic/claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.308,"output":0.924,"cache_read":0.0098}},"deepseek/deepseek-v4-flash-0423":{"id":"deepseek/deepseek-v4-flash-0423","name":"DeepSeek V4 Flash 0423","description":"Initial DeepSeek V4 Flash snapshot for economical reasoning, coding, and million-token agent workloads","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.19,"output":0.51,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.21,"output":0.84,"cache_read":0.0042}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"status":"beta","cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.924,"output":2.772,"cache_read":0.0308}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.15}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.29,"output":0.43,"cache_read":0.06}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":2.2,"cache_read":0.11}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.072,"output":0.4,"cache_read":0.01}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":4,"output":12,"cache_read":0.4}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"Grok 4.1 Fast","description":"xAI's fast agentic tool-calling model with a 2M context window; non-reasoning variant for low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.5}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.06,"output":0.56,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-1-6-flash":{"id":"volcengine/doubao-seed-1-6-flash","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.03,"output":0.22,"cache_read":0.0043}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.13,"output":0.76,"cache_read":0.03,"cache_write":0.0024}},"volcengine/doubao-seed-1-8":{"id":"volcengine/doubao-seed-1-8","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"volcengine/doubao-seed-1-6":{"id":"volcengine/doubao-seed-1-6","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"volcengine/doubao-seed-1-6-vision":{"id":"volcengine/doubao-seed-1-6-vision","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.12,"output":1.15,"cache_read":0.023}},"volcengine/doubao-seed-2.1-turbo":{"id":"volcengine/doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3536,"output":1.7696,"cache_read":0.068,"cache_write":0.0019}},"volcengine/doubao-seed-2.1-pro":{"id":"volcengine/doubao-seed-2.1-pro","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.7072,"output":3.536,"cache_read":0.1416,"cache_write":0.002}},"volcengine/doubao-seed-character":{"id":"volcengine/doubao-seed-character","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.177,"output":0.884,"cache_read":0.024,"cache_write":0.0025}},"volcengine/doubao-seed-evolving":{"id":"volcengine/doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.884,"output":4.42,"cache_read":0.177,"cache_write":0.0025}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083,"input_audio":3}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":1,"input_audio":1}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":0.75}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":1,"input_audio":0.3}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1,"input_audio":0.5}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax-M2.5 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax-M2.1 Lightning","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/m2-her":{"id":"minimax/m2-her","name":"MiniMax-M2 Her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.022,"output":0.22,"cache_read":0.0043,"cache_write":0.027}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.5,"output":1.71,"cache_read":0.043,"cache_write":0.63}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8000},"cost":{"input":0.35,"output":1.38,"cache_read":0.069}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.36,"output":1.43,"cache_read":0.072}},"qwen/qwen-turbo":{"id":"qwen/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.043,"output":0.09,"cache_read":0.0086}},"qwen/qwen-vl-max":{"id":"qwen/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.23,"output":0.58,"cache_read":0.023}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.29,"cache_read":0.29}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":2.5,"cache_read":0.06,"cache_write":0.27}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.8,"output":9,"cache_read":0.2,"cache_write":1}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11,"output":0.39,"cache_read":0.011,"cache_write":0.14}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.43,"output":2.57}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.71,"output":5.14,"cache_read":0.17,"cache_write":2.14}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":12,"cache_read":0.2}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":24,"output":144}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.16,"output":1,"cache_read":0.016}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":8,"cache_read":1}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.6,"cache_read":0.024}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.04,"output":0.32,"cache_read":0.008}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.12,"output":0.48,"cache_read":0.06}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.6,"output":3.6,"cache_read":0.06}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.4,"output":11.2,"cache_read":0.144}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.6,"output":6.4,"cache_read":0.4}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.32,"output":1.28,"cache_read":0.08}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.08,"output":0.4,"cache_read":0.008,"cache_write":0.1}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.6,"cache_read":0.024}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1,"output":8,"cache_read":0.104}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.6,"output":8,"cache_read":0.16,"cache_write":2}}}},"neon":{"id":"neon","env":["NEON_AI_GATEWAY_BASE_URL","NEON_AI_GATEWAY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"${NEON_AI_GATEWAY_BASE_URL}/v1","name":"Neon","doc":"https://neon.com/docs","models":{"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.3}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5-5-pro":{"id":"gpt-5-5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":10000},"cost":{"input":0.15,"output":1.2}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.07,"output":0.3}},"gpt-5-4":{"id":"gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":524288},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2,"output":6,"cache_read":0.5}},"qwen35-122b-a10b":{"id":"qwen35-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":25000},"cost":{"input":0.22,"output":2.2}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"meta-llama-3-3-70b-instruct":{"id":"meta-llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gpt-5-2":{"id":"gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemma-3-12b":{"id":"gemma-3-12b","name":"Gemma 3 12B","description":"Google's open-weight Gemma 3 vision-language model for text and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.5}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gpt-5-1":{"id":"gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.15,"output":0.6}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"meta-llama-3-1-8b-instruct":{"id":"meta-llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Meta's compact open-weight Llama 3.1 model for fast, low-cost text generation","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.45}}}},"aihubmix":{"id":"aihubmix","env":["AIHUBMIX_API_KEY"],"npm":"@aihubmix/ai-sdk-provider","name":"AIHubMix","doc":"https://docs.aihubmix.com","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.48,"output":0.96,"cache_read":0.00384}},"coding-minimax-m2.7-free":{"id":"coding-minimax-m2.7-free","name":"Coding MiniMax M2.7 (Free)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0,"output":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675,"cache_read":0.165}},"deep-deepseek-v4-pro":{"id":"deep-deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.478,"output":0.956,"cache_read":0.004302}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2.2,"output":6.6,"cache_read":0.55,"tiers":[{"input":4.4,"output":13.2,"cache_read":1.1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.4,"output":13.2,"cache_read":1.1}}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM 5 Vision Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.7042,"output":3.09848,"cache_read":0.169008}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.11268,"output":0.39438,"cache_read":0.02817}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.142,"output":0.284,"cache_read":0.0284}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Doubao Seed 2.0 Lite 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.08,"output":0.51,"cache_read":0.01692,"input_audio":1.269,"tiers":[{"input":0.13,"output":0.76,"cache_read":0.02536,"input_audio":1.902,"tier":{"type":"context","size":32000}},{"input":0.25,"output":1.52,"cache_read":0.05072,"input_audio":3.804,"tier":{"type":"context","size":128000}}]}},"claude-sonnet-4-6-think":{"id":"claude-sonnet-4-6-think","name":"Claude Sonnet 4.6 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"deepseek-v4-flash-0731-fast":{"id":"deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731 Fast","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.28,"output":1.4,"cache_read":0.07}},"hy3-preview":{"id":"hy3-preview","name":"Hy3 Preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.17,"output":0.566661,"cache_read":0.051}},"alicloud-deepseek-v4-pro":{"id":"alicloud-deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.69,"output":3.38,"cache_read":0.13}},"zai-glm-5.1":{"id":"zai-glm-5.1","name":"GLM-5.1 (Z.ai)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.845,"output":3.38,"cache_read":0.183112}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.155,"output":0.62,"cache_read":0.0031}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"coding-minimax-m2.7-highspeed":{"id":"coding-minimax-m2.7-highspeed","name":"Coding MiniMax M2.7 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.155,"output":0.62,"cache_read":0.0031}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":7.999,"cache_read":0.32167}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":64000},"cost":{"input":0.0282,"output":0.1128,"cache_read":0.00564,"cache_write":0.03525}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-opus-4-8-think":{"id":"claude-opus-4-8-think","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"doubao-seed-2-0-pro":{"id":"doubao-seed-2-0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"xiaomi-mimo-v2.5":{"id":"xiaomi-mimo-v2.5","name":"Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.088,"tiers":[{"input":0.88,"output":4.4,"cache_read":0.176,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.88,"output":4.4,"cache_read":0.176}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":1.08,"output":3.0888,"cache_read":0.054}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.7746,"output":3.0984,"cache_read":0.015492}},"mimo-v2.6-pro-ultraspeed":{"id":"mimo-v2.6-pro-ultraspeed","name":"MiMo-V2.6-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.8,"output":9.6,"cache_read":0.0384}},"doubao-seed-2-0-code-preview":{"id":"doubao-seed-2-0-code-preview","name":"Doubao Seed 2.0 Code Preview","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"qwen3.8-omni-flash":{"id":"qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1126,"output":0.380025,"cache_read":0.014075,"cache_write":0.175937}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Doubao Seed 2.0 Mini 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.03,"output":0.28,"cache_read":0.00564,"input_audio":0.423,"tiers":[{"input":0.06,"output":0.56,"cache_read":0.01128,"input_audio":0.846,"tier":{"type":"context","size":32000}},{"input":0.11,"output":1.13,"cache_read":0.02256,"input_audio":1.692,"tier":{"type":"context","size":128000}}]}},"xiaomi-mimo-v2.5-pro-free":{"id":"xiaomi-mimo-v2.5-pro-free","name":"Xiaomi MiMo-V2.5-Pro (free)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.499999,"cache_read":0.03}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"interleaved":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"coding-minimax-m2.7":{"id":"coding-minimax-m2.7","name":"Coding MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":1.5}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"deep-deepseek-v4-flash":{"id":"deep-deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepSeek)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.154,"output":0.308,"cache_read":0.0308}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.17,"output":1.01,"cache_read":0.0169,"cache_write":0.21125,"tiers":[{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845}}},"coding-glm-5.1":{"id":"coding-glm-5.1","name":"Coding GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.22,"cache_read":0.013}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"xiaomi-mimo-v2.5-free":{"id":"xiaomi-mimo-v2.5-free","name":"Xiaomi MiMo-V2.5 (free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.6918,"output":2.0754,"cache_read":0.023058}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.375,"output":4.675}},"gemini-3.1-flash-lite-image":{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.155,"output":0.31,"cache_read":0.0031}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.288,"output":1.152}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1126,"output":0.380025,"cache_read":0.014075,"cache_write":0.175937}},"claude-opus-4-6-think":{"id":"claude-opus-4-6-think","name":"Claude Opus 4.6 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"coding-xiaomi-mimo-v2.5":{"id":"coding-xiaomi-mimo-v2.5","name":"Coding Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.08,"output":0.4,"cache_read":0.016,"tiers":[{"input":0.16,"output":0.8,"cache_read":0.032,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.16,"output":0.8,"cache_read":0.032}}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"ox-alpha":{"id":"ox-alpha","name":"Ox Alpha","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.27,"output":7.61,"cache_read":0.1268,"cache_write":1.585,"tiers":[{"input":2.11,"output":12.67,"cache_read":0.2112,"cache_write":2.64,"tier":{"type":"context","size":128000}}]}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.1562,"output":0.6248,"cache_read":0.03905}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"claude-opus-4-7-think":{"id":"claude-opus-4-7-think","name":"Claude Opus 4.7 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"alicloud-deepseek-v4-flash":{"id":"alicloud-deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"coding-xiaomi-mimo-v2.5-pro":{"id":"coding-xiaomi-mimo-v2.5-pro","name":"Coding Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.2,"output":0.6,"cache_read":0.04,"tiers":[{"input":0.4,"output":1.2,"cache_read":0.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.4,"output":1.2,"cache_read":0.08}}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.22,"output":1.32,"cache_read":0.044}},"alicloud-glm-5.1":{"id":"alicloud-glm-5.1","name":"GLM-5.1 (Alibaba Cloud)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.84,"output":3.38,"cache_read":0.169,"cache_write":1.05625}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"coding-glm-5.1-free":{"id":"coding-glm-5.1-free","name":"Coding GLM 5.1 (free)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.28,"output":1.69,"cache_read":0.0282,"cache_write":0.3525,"tiers":[{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.845,"output":2.535,"cache_read":0.04225}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":3.9995,"cache_read":0.160835}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xiaomi-mimo-v2.5-pro":{"id":"xiaomi-mimo-v2.5-pro","name":"Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.1,"output":3.3,"cache_read":0.22,"tiers":[{"input":2.2,"output":6.6,"cache_read":0.44,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.2,"output":6.6,"cache_read":0.44}}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.282,"output":1.128,"cache_read":0.0564,"cache_write":0.3525}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}}}},"merge-gateway":{"id":"merge-gateway","env":["MERGE_GATEWAY_API_KEY"],"npm":"merge-gateway-ai-sdk-provider","api":"https://api-gateway.merge.dev/v1/ai-sdk","name":"Merge Gateway","doc":"https://docs.merge.dev/merge-gateway","models":{"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.015,"output":0.05,"cache_read":0.003}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"Glm 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11,"cache_write":0}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":50000}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.05,"output":3.3,"cache_read":0.195}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.7,"output":2.2,"cache_read":0.13}},"anthropic/claude-3-7-sonnet-20250219":{"id":"anthropic/claude-3-7-sonnet-20250219","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5-5":{"id":"anthropic/claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-opus-4-20250514":{"id":"anthropic/claude-opus-4-20250514","name":"Claude Opus 4 (20250514)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-1-20250805":{"id":"anthropic/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (20251101)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-4-20250514":{"id":"anthropic/claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R 08-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A 03-2025","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B 12-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+ 08-2024","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.035,"output":0.07,"cache_read":0.007}},"deepseek/deepseek-v4-flash-0423":{"id":"deepseek/deepseek-v4-flash-0423","name":"DeepSeek V4 Flash 0423","description":"Initial DeepSeek V4 Flash snapshot for economical reasoning, coding, and million-token agent workloads","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.139,"output":0.278}},"deepseek/deepseek-v4-flash-0731-fast":{"id":"deepseek/deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v3":{"id":"deepseek/deepseek-v3","name":"DeepSeek V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.003625}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":41000},"cost":{"input":0.5,"output":1.5}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":1.35,"output":5.4}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":0.28,"output":0.45,"cache_read":0.14}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.035,"output":0.07,"cache_read":0.007}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":32000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":0.5,"cache_read":0.11}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.22,"output":0.22}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.99,"output":0.99}},"bytedance/dola-seed-2.0-code-preview":{"id":"bytedance/dola-seed-2.0-code-preview","name":"Dola Seed 2.0 Code (preview)","description":"Preview coding model for repository understanding, refactors, and engineering tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"bytedance/dola-seed-2.0-code":{"id":"bytedance/dola-seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4}},"bytedance/dola-seed-2.0-mini":{"id":"bytedance/dola-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Low-cost Seed model for general chat, extraction, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.4}},"bytedance/dola-seed-2.0-lite":{"id":"bytedance/dola-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Efficient Seed model for general chat, analysis, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-28","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":2}},"bytedance/dola-seed-2.0-pro":{"id":"bytedance/dola-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Higher-capability Seed model for complex chat, analysis, and production tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B It","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.14,"output":0.4}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.08,"output":0.45,"cache_read":0.04}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.13,"output":0.4}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Gemini 3.1 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.09,"output":0.29}},"google/gemini-2.5-computer-use-preview-10-2025":{"id":"google/gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview (10-2025)","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":1.25,"output":10}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":2,"output":12}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.08}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":4096},"cost":{"input":0.15,"output":0}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"writer/palmyra-x4":{"id":"writer/palmyra-x4","name":"Palmyra X4","description":"Enterprise language model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-10-09","last_updated":"2024-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":2.5,"output":10}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"Enterprise multimodal model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.6,"output":6}},"xai/grok-4.7":{"id":"xai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.6,"output":2.5}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0,"output":0}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nemotron Nano 9B","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.06,"output":0.23}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/mistral-large-2411":{"id":"mistral/mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"mistral/devstral-medium-2507":{"id":"mistral/devstral-medium-2507","name":"Devstral Medium","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/devstral-small-2507":{"id":"mistral/devstral-small-2507","name":"Devstral Small","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral/pixtral-large-latest":{"id":"mistral/pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.022,"output":0.216,"cache_read":0.0044}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.029,"output":0.287,"cache_read":0.0058}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3-VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.15785}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.825,"output":2.4755,"cache_read":0.165}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3-VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":2.867,"cache_read":0.0574}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":1010000},"cost":{"input":2.5,"output":6.25,"cache_read":0.5}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.143,"output":1.434,"cache_read":0.0286}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.086,"output":0.688,"cache_read":0.0172}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.6,"cache_read":0.05}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.8,"cache_read":0.075}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.688,"cache_read":0.023}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.287,"cache_read":0.023}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434,"cache_read":0.0718}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":1.8}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.0574}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.115,"output":0.917,"cache_read":0.023}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485,"cache_read":0.0496}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.165,"output":0.99,"cache_read":0.033}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.574,"output":2.294,"cache_read":0.1148}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.057,"output":0.459,"cache_read":0.020357}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":1.076,"cache_read":0.0216}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.172,"output":1.032,"cache_read":0.0344}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.289,"output":2.4}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.276,"output":1.651,"cache_read":0.0552}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.15,"output":0.6}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.2,"cache_read":0.02}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.07,"output":0.2,"cache_read":0,"cache_write":0}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"openai/gpt-5.3-chat-latest":{"id":"openai/gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.36}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":2.9,"output":14,"cache_read":0.3}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"opper":{"id":"opper","env":["OPPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.opper.ai/v3/compat","name":"Opper","doc":"https://opper.ai/models","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.46488,"output":2.44062}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.5811,"output":3}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.07}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.5811,"output":2.3244}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125,"tiers":[{"input":6.6,"output":24.75,"cache_read":0.66,"cache_write":8.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6.6,"output":24.75,"cache_read":0.66,"cache_write":8.25}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.4,"output":2}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.11622,"output":0.488124}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.69732,"output":2.78928}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":524288}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5811,"output":2.44062}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.62708,"output":5.811}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":1.78812,"output":3.57624}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":1.1622,"output":4.88124}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.75,"output":4.6488,"cache_read":0.44}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.25,"output":0.66}}}},"nvidia":{"id":"nvidia","env":["NVIDIA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://integrate.api.nvidia.com/v1","name":"Nvidia","doc":"https://docs.api.nvidia.com/nim/","models":{"baai/bge-m3":{"id":"baai/bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0,"output":0}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"abacusai/dracarys-llama-3.1-70b-instruct":{"id":"abacusai/dracarys-llama-3.1-70b-instruct","name":"dracarys-llama-3.1-70b-instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-11","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-schnell":{"id":"black-forest-labs/flux_1-schnell","name":"FLUX.1-schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-07","release_date":"2024-08-01","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":77,"input":77,"output":0},"cost":{"input":0,"output":0}},"black-forest-labs/flux.1-dev":{"id":"black-forest-labs/flux.1-dev","name":"FLUX.1-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-kontext-dev":{"id":"black-forest-labs/flux_1-kontext-dev","name":"FLUX.1-Kontext-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"black-forest-labs/flux_2-klein-4b":{"id":"black-forest-labs/flux_2-klein-4b","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-14","last_updated":"2026-01-31","modalities":{"input":["image","text"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"meta/llama-guard-4-12b":{"id":"meta/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"meta/esmfold":{"id":"meta/esmfold","name":"esmfold","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-15","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-26","last_updated":"2024-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.2-90b-vision-instruct":{"id":"meta/llama-3.2-90b-vision-instruct","name":"Llama-3.2-90B-Vision-Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-4-maverick-17b-128e-instruct":{"id":"meta/llama-4-maverick-17b-128e-instruct","name":"Llama 4 Maverick 17b 128e Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-02","release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11b Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0,"output":0}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/esm2-650m":{"id":"meta/esm2-650m","name":"esm2-650m","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-29","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"upstage/solar-10.7b-instruct":{"id":"upstage/solar-10.7b-instruct","name":"solar-10.7b-instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-05","last_updated":"2025-04-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"bytedance/seed-oss-36b-instruct":{"id":"bytedance/seed-oss-36b-instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma-4-31B-IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-2-2b-it":{"id":"google/gemma-2-2b-it","name":"Gemma 2 2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3n-e2b-it":{"id":"google/gemma-3n-e2b-it","name":"Gemma 3n E2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-12","last_updated":"2025-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/google-paligemma":{"id":"google/google-paligemma","name":"paligemma","description":"Gemini multimodal model for text, image, audio, video, and document tasks","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-14","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"google/gemma-3n-e4b-it":{"id":"google/gemma-3n-e4b-it","name":"Gemma 3n E4b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"minimaxai/minimax-m2.7":{"id":"minimaxai/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-pro-0813":{"id":"deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"stepfun-ai/step-3.7-flash":{"id":"stepfun-ai/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-nemotron":{"id":"mistralai/mistral-nemotron","name":"mistral-nemotron","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mistral-7b-instruct-v0.3":{"id":"mistralai/mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0,"output":0}},"mistralai/mistral-large-3-675b-instruct-2512":{"id":"mistralai/mistral-large-3-675b-instruct-2512","name":"Mistral Large 3 675B Instruct 2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"mistralai/magistral-small-2506":{"id":"mistralai/magistral-small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0,"output":0}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"mistral-small-4-119b-2603","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x7b-instruct":{"id":"mistralai/mixtral-8x7b-instruct","name":"Mistral: Mixtral 8x7B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2023-12-10","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3.5-128b":{"id":"mistralai/mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B Instruct 2512","description":"Compact Mistral VLM for chat and instruction-based workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":13108},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3-instruct":{"id":"mistralai/mistral-medium-3-instruct","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0,"output":0}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"moonshotai/kimi-k2-instruct-0905":{"id":"moonshotai/kimi-k2-instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"sarvamai/sarvam-m":{"id":"sarvamai/sarvam-m","name":"sarvam-m","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nvidia-nemotron-nano-9b-v2":{"id":"nvidia/nvidia-nemotron-nano-9b-v2","name":"nvidia-nemotron-nano-9b-v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-ultra-253b-v1":{"id":"nvidia/llama-3.1-nemotron-ultra-253b-v1","name":"Llama 3.1 Nemotron Ultra 253B","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"nvidia/sparsedrive":{"id":"nvidia/sparsedrive","name":"sparsedrive","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"nemotron-3-nano-30b-a3b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"nvidia/llama-nemotron-embed-vl-1b-v2":{"id":"nvidia/llama-nemotron-embed-vl-1b-v2","name":"llama-nemotron-embed-vl-1b-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/nv-embedcode-7b-v1":{"id":"nvidia/nv-embedcode-7b-v1","name":"nv-embedcode-7b-v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-17","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/nemotron-voicechat":{"id":"nvidia/nemotron-voicechat","name":"nemotron-voicechat","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/bevformer":{"id":"nvidia/bevformer","name":"bevformer","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer1-7b":{"id":"nvidia/cosmos-transfer1-7b","name":"cosmos-transfer1-7b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-13","last_updated":"2025-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/cosmos-predict1-5b":{"id":"nvidia/cosmos-predict1-5b","name":"cosmos-predict1-5b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-vl-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-vl-8b-v1","name":"Llama 3.1 Nemotron Nano VL 8B v1","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-10","last_updated":"2025-04-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"nvidia/studiovoice":{"id":"nvidia/studiovoice","name":"studiovoice","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-03","last_updated":"2025-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-content-safety-reasoning-4b":{"id":"nvidia/nemotron-content-safety-reasoning-4b","name":"nemotron-content-safety-reasoning-4b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-safety-guard-8b-v3":{"id":"nvidia/llama-3.1-nemotron-safety-guard-8b-v3","name":"llama-3.1-nemotron-safety-guard-8b-v3","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nv-embed-v1":{"id":"nvidia/nv-embed-v1","name":"nv-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-07","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/synthetic-video-detector":{"id":"nvidia/synthetic-video-detector","name":"synthetic-video-detector","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer2_5-2b":{"id":"nvidia/cosmos-transfer2_5-2b","name":"cosmos-transfer2.5-2b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.15}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":-1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-8b-v1","name":"Llama 3.1 Nemotron Nano 8B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"nvidia/active-speaker-detection":{"id":"nvidia/active-speaker-detection","name":"Active Speaker Detection","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/streampetr":{"id":"nvidia/streampetr","name":"streampetr","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-nemotron-rerank-vl-1b-v2":{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2","name":"llama-nemotron-rerank-vl-1b-v2","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-mini-4b-instruct":{"id":"nvidia/nemotron-mini-4b-instruct","name":"nemotron-mini-4b-instruct","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-08-21","last_updated":"2024-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-reason2-8b":{"id":"nvidia/cosmos-reason2-8b","name":"Cosmos Reason2 8B","description":"Vision language model for physical-world understanding with structured reasoning on video and images","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-content-safety":{"id":"nvidia/nemotron-3-content-safety","name":"nemotron-3-content-safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1","name":"Llama 3.3 Nemotron Super 49B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/magpie-tts-zeroshot":{"id":"nvidia/magpie-tts-zeroshot","name":"magpie-tts-zeroshot","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-06-12","modalities":{"input":["text","audio"],"output":["audio"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/riva-translate-4b-instruct-v1.1":{"id":"nvidia/riva-translate-4b-instruct-v1.1","name":"riva-translate-4b-instruct-v1_1","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nemotron Nano 12B v2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1.5":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/llama-3_2-nemoretriever-300m-embed-v1":{"id":"nvidia/llama-3_2-nemoretriever-300m-embed-v1","name":"llama-3_2-nemoretriever-300m-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-24","last_updated":"2025-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/gliner-pii":{"id":"nvidia/gliner-pii","name":"gliner-pii","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/usdcode":{"id":"nvidia/usdcode","name":"usdcode","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/usdvalidate":{"id":"nvidia/usdvalidate","name":"usdvalidate","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-24","last_updated":"2025-01-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-70b-instruct":{"id":"nvidia/llama-3.1-nemotron-70b-instruct","name":"Llama 3.1 Nemotron 70B Instruct","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/rerank-qa-mistral-4b":{"id":"nvidia/rerank-qa-mistral-4b","name":"rerank-qa-mistral-4b","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-17","last_updated":"2025-01-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"qwen/qwen-image-edit":{"id":"qwen/qwen-image-edit","name":"Qwen Image Edit","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":66536},"cost":{"input":0,"output":0}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"qwen/qwen2.5-coder-32b-instruct":{"id":"qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32b Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-06","last_updated":"2024-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"qwen/qwen-image":{"id":"qwen/qwen-image","name":"Qwen Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS-120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-04","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"microsoft/phi-4-mini-instruct":{"id":"microsoft/phi-4-mini-instruct","name":"Phi-4-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"microsoft/phi-4-multimodal-instruct":{"id":"microsoft/phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0,"output":0}}}},"pioneer":{"id":"pioneer","env":["PIONEER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pioneer.ai/v1","name":"Pioneer","doc":"https://agent.pioneer.ai/llms.txt","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25,"cache_write":2.5}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"magistral-medium":{"id":"magistral-medium","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":2,"output":5,"cache_read":2,"cache_write":2}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025,"cache_write":0.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-3-7-sonnet-latest":{"id":"claude-3-7-sonnet-latest","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05,"cache_write":0.1}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.5,"output":7.5,"cache_read":1.5,"cache_write":1.5}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005,"cache_write":0.05}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"cache_write":1.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":131072},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.083333}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_read":0.0375,"cache_write":0.234375}},"mistral-large-3":{"id":"mistral-large-3","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"devstral-small-2":{"id":"devstral-small-2","name":"Devstral Small 2","description":"Compact multimodal coding model for repository exploration, file editing, and software agents","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.1,"cache_write":0.1}},"devstral-2":{"id":"devstral-2","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.4,"cache_write":0.4}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":1,"cache_write":2}},"mistral-medium":{"id":"mistral-medium","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.4,"output":2,"cache_read":0.4,"cache_write":0.4}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.04,"output":6.24,"cache_read":0.208,"cache_write":1.3}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.2,"cache_write":0.4}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5-fast":{"id":"claude-opus-5-fast","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.325,"output":1.95,"cache_read":0.065,"cache_write":0.40625}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"ministral-14b":{"id":"ministral-14b","name":"Ministral 14B","description":"Compact multimodal Mistral model for local assistants, edge agents, and efficient tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03,"cache_write":0.25}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01,"cache_write":0.1}},"meta-llama/Llama-3.2-1B-Instruct":{"id":"meta-llama/Llama-3.2-1B-Instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":60000},"cost":{"input":0.1,"output":0.201,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-3B-Instruct":{"id":"meta-llama/Llama-3.2-3B-Instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":80000},"cost":{"input":0.1,"output":0.335,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-06-30","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"meta-llama/Llama-3.2-3B":{"id":"meta-llama/Llama-3.2-3B","name":"Llama-3.2-3B","description":"Small open Llama base model for lightweight text generation and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-1B":{"id":"meta-llama/Llama-3.2-1B","name":"Llama-3.2-1B","description":"Compact open Llama base model for lightweight and on-device use","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0.435}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0.14}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":1.2,"cache_read":0.1,"cache_write":0.5}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15,"cache_write":1.25}},"HuggingFaceTB/SmolLM3-3B-Base":{"id":"HuggingFaceTB/SmolLM3-3B-Base","name":"SmolLM3 3B Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"google/diffusiongemma-26B-A4B-it":{"id":"google/diffusiongemma-26B-A4B-it","name":"DiffusionGemma 26B-A4B IT","description":"Gemini model for general assistance, reasoning, and multimodal workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-12B-it":{"id":"google/gemma-4-12B-it","name":"Gemma 4 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.25,"cache_read":0.25,"cache_write":0.25}},"google/gemma-4-E2B-it":{"id":"google/gemma-4-E2B-it","name":"Gemma 4 E2B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"google/gemma-3-4b-pt":{"id":"google/gemma-3-4b-pt","name":"Gemma 3 4B (Pretrained)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-02-28","last_updated":"2025-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3-4B-Instruct-2507":{"id":"Qwen/Qwen3-4B-Instruct-2507","name":"Qwen3 4B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen2.5-Coder-0.5B":{"id":"Qwen/Qwen2.5-Coder-0.5B","name":"Qwen2.5-Coder-0.5B","description":"Tiny open Qwen code model for lightweight completion and on-device coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3-1.7B-Base":{"id":"Qwen/Qwen3-1.7B-Base","name":"Qwen3 1.7B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":1,"cache_read":0.028,"cache_write":0.175}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.2,"output":1.2,"cache_read":1.2,"cache_write":1.2}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-03-31","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3-4B-Base":{"id":"Qwen/Qwen3-4B-Base","name":"Qwen3 4B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.3,"cache_write":0.3}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.6,"output":0.6,"cache_read":0.6,"cache_write":0.6}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2 24B A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-01-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12,"cache_read":0.03,"cache_write":0.03}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072},"cost":{"input":0.56,"output":1.68,"cache_read":0.56,"cache_write":0.56}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.0197,"cache_write":0.1}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625,"cache_write":0.435}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.27,"output":1.12,"cache_read":0.135,"cache_write":0.27}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.3}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.279,"output":1.2,"cache_read":0.279,"cache_write":0.279}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}},"mistralai/Pixtral-12B-2409":{"id":"mistralai/Pixtral-12B-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03,"cache_read":0.02,"cache_write":0.02}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small","description":"Open Mistral reasoning model for transparent step-by-step problem solving","family":"magistral","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"mistralai/Ministral-8B-Instruct-2410":{"id":"mistralai/Ministral-8B-Instruct-2410","name":"Ministral 8B Instruct","description":"Efficient open Mistral edge model for on-device chat and function calling","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Codestral-22B-v0.1":{"id":"mistralai/Codestral-22B-v0.1","name":"Codestral-22B-v0.1","description":"Open Mistral code model for fill-in-the-middle and 80+ programming languages","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-29","last_updated":"2024-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.3,"output":0.9,"cache_read":0.3,"cache_write":0.3}},"mistralai/Mistral-7B-Instruct-v0.3":{"id":"mistralai/Mistral-7B-Instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2023-04-30","last_updated":"2023-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.34,"cache_write":0.95}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"moonshotai/Kimi-K3-Fast":{"id":"moonshotai/Kimi-K3-Fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45,"cache_write":4.5}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19,"cache_write":0.95}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182,"cache_write":0.98}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1040000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":1.4}},"zai-org/GLM-5.2-Fast":{"id":"zai-org/GLM-5.2-Fast","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.1,"output":6.6,"cache_read":0.21,"cache_write":2.1}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.05,"cache_write":0.05}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":2.5,"cache_read":0.15,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.09,"output":0.45,"cache_read":0.09,"cache_write":0.09}},"pioneer/auto":{"id":"pioneer/auto","name":"Pioneer Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2025-06-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":4096}},"fastino/gliner2-multi-v1":{"id":"fastino/gliner2-multi-v1","name":"GLiNER2 Multi","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-privacy-filter-PII-multi":{"id":"fastino/gliner2-privacy-filter-PII-multi","name":"GLiNER2 Privacy Filter PII (Multi)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2.5-multi-v1":{"id":"fastino/gliner2.5-multi-v1","name":"GLiNER 2.5 Multi","description":"Multilingual boundary NER and span extraction; non-trainable encoder.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-24","last_updated":"2026-08-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliguard-LLMGuardrails-300M":{"id":"fastino/gliguard-LLMGuardrails-300M","name":"GLiGuard LLM Guardrails 300M","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-large-v1":{"id":"fastino/gliner2-large-v1","name":"GLiNER2 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-base-v1":{"id":"fastino/gliner2-base-v1","name":"GLiNER2 Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliguard-PII-multi":{"id":"fastino/gliguard-PII-multi","name":"GLiNER2-Guardrails-PII-Multi","description":"A 300M-parameter multilingual model that runs LLM safety moderation and PII detection in a single forward pass.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-multi-large-v1":{"id":"fastino/gliner2-multi-large-v1","name":"GLiNER2 Multi Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.035,"cache_write":0.07}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}}}},"xiaomi":{"id":"xiaomi","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.xiaomimimo.com/v1","name":"Xiaomi","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"MiMo Pro model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"MiMo-V2-Flash","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo-V2-Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.6-pro-ultraspeed":{"id":"mimo-v2.6-pro-ultraspeed","name":"MiMo-V2.6-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.35,"output":8.7,"cache_read":0.036}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5-pro-ultraspeed":{"id":"mimo-v2.5-pro-ultraspeed","name":"MiMo-V2.5-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-06-08","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":1.305,"output":2.61,"cache_read":0.0108}}}},"xiaomi-token-plan-sgp":{"id":"xiaomi-token-plan-sgp","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-sgp.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Singapore)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"minimax":{"id":"minimax","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax (minimax.io)","doc":"https://platform.minimax.io/docs/guides/quickstart","models":{"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}}}},"github-copilot":{"id":"github-copilot","env":["GITHUB_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.githubcopilot.com","name":"GitHub Copilot","doc":"https://docs.github.com/en/copilot","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"claude-opus-4.7":{"id":"claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":264000,"input":128000,"output":64000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":64000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":32000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"mai-code-1-flash-picker":{"id":"mai-code-1-flash-picker","name":"MAI-Code-1-Flash","description":"Microsoft coding model built for fast, efficient assistance in everyday developer workflows","family":"mai","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-06-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"mai-code-1.1-flash":{"id":"mai-code-1.1-flash","name":"MAI-Code-1.1-Flash","description":"Microsoft coding model with native vision support, optimized for fast and efficient software development","family":"mai","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":136000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":24000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":128000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"claude-sonnet-4.6":{"id":"claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"claude-opus-5.5":{"id":"claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":224000,"output":32000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}}}},"inferx":{"id":"inferx","env":["INFERX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://model.inferx.net/endpoints/v1","name":"InferX","doc":"https://model.inferx.net/endpoints","models":{"gemma-4-31B-it-fp8":{"id":"gemma-4-31B-it-fp8","name":"Gemma 4 31B IT FP8","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-fp8-no-thinking":{"id":"Qwen3.6-35B-A3B-fp8-no-thinking","name":"Qwen3.6-35B-A3B-fp8-no-thinking","description":"Qwen3.6-35B-A3B-fp8 disable thinking","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Devstral-2-123B-Instruct-2512-int4-AutoRound":{"id":"Devstral-2-123B-Instruct-2512-int4-AutoRound","name":"Devstral-2-123B-Instruct-2512-int4-AutoRound","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"Qwen3.6-27B-FP8":{"id":"Qwen3.6-27B-FP8","name":"Qwen3.6 27B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8":{"id":"Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8-no-thinking":{"id":"Qwen3-Coder-Next-FP8-no-thinking","name":"Qwen3-Coder-Next-FP8-no-thinking","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":65536},"cost":{"input":0,"output":0}},"mimo-v25":{"id":"mimo-v25","name":"mimo-v25","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}},"Qwen3-Embedding-8B":{"id":"Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":0},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-FP8":{"id":"Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Ornith-1.0-35B-FP8":{"id":"Ornith-1.0-35B-FP8","name":"Ornith-1.0-35B-FP8","description":"Large coding-reasoning model for agentic software tasks and RL search","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-25","last_updated":"2026-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"Agents-A1":{"id":"Agents-A1","name":"Agents-A1","description":"35B MoE agentic model built for long-horizon search, engineering, and scientific reasoning tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}}}},"opencode-go":{"id":"opencode-go","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/go/v1","name":"OpenCode Go","doc":"https://opencode.ai/docs/go","models":{"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.7-max","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo V2 Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-omni","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2,"cache_read":0.08}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter multimodal flagship for coding, professional work, and long-horizon agentic workflows","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Legacy model retained for compatibility with older integrations","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"status":"deprecated","cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"ox-alpha-free":{"id":"ox-alpha-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.6,"output":3,"cache_read":0.1}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo V2 Pro","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"status":"deprecated","cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"space-bunny-free":{"id":"space-bunny-free","name":"Space Bunny Free","description":"Anonymous preview reasoning model for coding, agentic tasks, tool use, and multimodal input","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":524288,"output":524288},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo-v2.5-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"omen-alpha":{"id":"omen-alpha","name":"Omen Alpha","description":"oH man anothEr aLPha ModEl","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":128000},"status":"deprecated","cost":{"input":0.2,"output":0.66,"cache_read":0.04}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro (New)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.7-plus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}}}},"openreason":{"id":"openreason","env":["OPENREASON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openreason.app/v1","name":"OpenReason","doc":"https://openreason.app/docs","models":{"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1371,"output":0.2743}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.0022,"output":4.22}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1055,"output":0.422}}}},"lmstudio":{"id":"lmstudio","env":["LMSTUDIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1234/v1","name":"LMStudio","doc":"https://lmstudio.ai/models","models":{"qwen/qwen3-30b-a3b-2507":{"id":"qwen/qwen3-30b-a3b-2507","name":"Qwen3 30B A3B 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"qwen/qwen3-coder-30b":{"id":"qwen/qwen3-coder-30b","name":"Qwen3 Coder 30B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"aki-io":{"id":"aki-io","env":["AKI_IO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://aki.io/v1","name":"AKI.IO","doc":"https://aki.io/docs/","models":{"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.2,"cache_read":0.1}},"deepseek-v4-flash-0731-284b":{"id":"deepseek-v4-flash-0731-284b","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":81920},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.15,"output":0.5}},"glm5.3-754b":{"id":"glm5.3-754b","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":81920},"cost":{"input":1,"output":3.5,"cache_read":0.25}},"mistral4-119b":{"id":"mistral4-119b","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.2,"output":0.6}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.55}},"gemma4-26b":{"id":"gemma4-26b","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.5}}}},"tensorx":{"id":"tensorx","env":["TENSORX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tensorx.ai/v1","name":"TensorX","doc":"https://docs.tensorx.ai/","models":{"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.06}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.5,"output":1.5,"cache_read":0.13}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":2,"output":4,"cache_read":0.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.75,"output":3.5,"cache_read":0.4375,"cache_write":2.185}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1-0528","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":8192},"cost":{"input":0.66,"output":2.6,"cache_read":0.165,"cache_write":0.825}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.3,"output":0.5,"cache_read":0.075,"cache_write":0.375}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1,"output":3.2,"cache_read":0.25,"cache_write":1.25}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.5,"output":4.5,"cache_read":0.375}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.4,"output":4.4,"cache_read":0.35,"cache_write":1.75}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":1.75,"output":4.5,"cache_read":0.44}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1,"output":4,"cache_read":0.25,"cache_write":1.25}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125,"cache_write":0.625}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.3125}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.075,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B-A22B-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":262144},"cost":{"input":0.072,"output":0.464,"cache_read":0.018,"cache_write":0.09}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":2.4,"cache_read":0.1}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":3.5,"cache_read":0.125,"cache_write":0.625}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.2,"cache_read":0.0375,"cache_write":0.1875}},"qwen/qwen3.8-flash-next":{"id":"qwen/qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}}}},"longcat":{"id":"longcat","env":["LONGCAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.longcat.chat/openai","name":"LongCat","doc":"https://longcat.chat/platform/docs/","models":{"LongCat-2.0":{"id":"LongCat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.75,"output":2.95,"cache_read":0.015}}}},"chutes":{"id":"chutes","env":["CHUTES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.chutes.ai/v1","name":"Chutes","doc":"https://llm.chutes.ai/v1/models","models":{"Nemotron-3-Nano-Omni-30B-TEE":{"id":"Nemotron-3-Nano-Omni-30B-TEE","name":"Nemotron 3 Nano Omni 30B TEE","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":0},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"unsloth/Mistral-Nemo-Instruct-2407-TEE":{"id":"unsloth/Mistral-Nemo-Instruct-2407-TEE","name":"Mistral Nemo Instruct 2407 TEE","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"google/gemma-4-31B-turbo-TEE":{"id":"google/gemma-4-31B-turbo-TEE","name":"gemma 4 31B turbo TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.12,"output":0.37,"cache_read":0.011999999999999997}},"Qwen/Qwen3-32B-TEE":{"id":"Qwen/Qwen3-32B-TEE","name":"Qwen3 32B TEE","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.104,"output":0.416,"cache_read":0.010399999999999998}},"Qwen/Qwen3.6-27B-TEE":{"id":"Qwen/Qwen3.6-27B-TEE","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2,"cache_read":0.029999999999999992}},"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE","name":"Qwen3 235B A22B Thinking 2507 TEE","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2989,"output":1.1957,"cache_read":0.029889999999999993}},"Qwen/Qwen3.8-27B-TEE":{"id":"Qwen/Qwen3.8-27B-TEE","name":"Qwen3.8 27B TEE","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.24,"output":2.2,"cache_read":0.023999999999999994}},"Qwen/Qwen3.5-397B-A17B-TEE":{"id":"Qwen/Qwen3.5-397B-A17B-TEE","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3,"cache_read":0.04499999999999999}},"deepseek-ai/DeepSeek-V4-Flash-0731-TEE":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731-TEE","name":"DeepSeek V4 Flash 0731 TEE","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.04399999999999999}},"deepseek-ai/DeepSeek-V3.2-TEE":{"id":"deepseek-ai/DeepSeek-V3.2-TEE","name":"DeepSeek V3.2 TEE","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":1,"cache_read":0.09999999999999998}},"moonshotai/Kimi-K3-TEE":{"id":"moonshotai/Kimi-K3-TEE","name":"Kimi K3 TEE","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":0.29999999999999993}},"moonshotai/Kimi-K2.6-TEE":{"id":"moonshotai/Kimi-K2.6-TEE","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65535},"cost":{"input":0.5,"output":2.85,"cache_read":0.04999999999999999}},"zai-org/GLM-5.2-TEE":{"id":"zai-org/GLM-5.2-TEE","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":3.95,"cache_read":0.12499999999999997}},"zai-org/GLM-5.1-TEE":{"id":"zai-org/GLM-5.1-TEE","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":65535},"cost":{"input":0.98,"output":3.08,"cache_read":0.09799999999999998}}}},"edenai":{"id":"edenai","env":["EDENAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.edenai.run/v3","name":"Eden AI","doc":"https://docs.edenai.co","models":{"deepinfra/nemotron-3-ultra-550b-a55b":{"id":"deepinfra/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Deep Infra)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/meta-models/Muse-Glimmer-30B":{"id":"deepinfra/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Deep Infra)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"deepinfra/tencent/Hy3":{"id":"deepinfra/tencent/Hy3","name":"Hy3 (Deep Infra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.13,"output":0.53,"cache_read":0.033}},"deepinfra/meta-llama/Llama-Guard-3-8B":{"id":"deepinfra/meta-llama/Llama-Guard-3-8B","name":"Llama-Guard-3-8B (Deep Infra)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.055,"output":0.055}},"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct":{"id":"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct","name":"Llama-3.2-11B-Vision-Instruct (Deep Infra)","description":"Open multimodal Llama model for image understanding, captioning, and visual QA","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.345,"output":0.345}},"deepinfra/meta-llama/Llama-3.3-70B-Instruct":{"id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (Deep Infra)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.1,"output":0.32}},"deepinfra/thinkingmachines/Inkling-Small":{"id":"deepinfra/thinkingmachines/Inkling-Small","name":"Inkling Small (Deep Infra)","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"deepinfra/thinkingmachines/Inkling":{"id":"deepinfra/thinkingmachines/Inkling","name":"Inkling (Deep Infra)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"deepinfra/google/gemma-3-27b-it":{"id":"deepinfra/google/gemma-3-27b-it","name":"Gemma 3 27B IT (Deep Infra)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.08,"output":0.16}},"deepinfra/google/gemma-3-12b-it":{"id":"deepinfra/google/gemma-3-12b-it","name":"Gemma 3 12B IT (Deep Infra)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.15}},"deepinfra/google/gemma-3-4b-it":{"id":"deepinfra/google/gemma-3-4b-it","name":"Gemma 3 4B IT (Deep Infra)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1}},"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Deep Infra)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepinfra/deepseek-ai/DeepSeek-R1":{"id":"deepinfra/deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1 (Deep Infra)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.4}},"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Deep Infra)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Deep Infra)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.2,"output":0.6,"cache_read":0.006}},"deepinfra/deepseek-ai/DeepSeek-V3":{"id":"deepinfra/deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3 (Deep Infra)","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepinfra/deepseek-ai/DeepSeek-V3-0324":{"id":"deepinfra/deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324 (Deep Infra)","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepinfra/stepfun-ai/Step-3.7-Flash":{"id":"deepinfra/stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash (Deep Infra)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepinfra/stepfun-ai/Step-3.5-Flash":{"id":"deepinfra/stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash (Deep Infra)","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"deepinfra/moonshotai/Kimi-K2.5":{"id":"deepinfra/moonshotai/Kimi-K2.5","name":"Kimi K2.5 (Deep Infra)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"deepinfra/zai-org/GLM-4.7-Flash":{"id":"deepinfra/zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash (Deep Infra)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B":{"id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B (Deep Infra)","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct":{"id":"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct","name":"Llama 3.1 Nemotron 70B Instruct (Deep Infra)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.6,"output":0.6}},"deepinfra/ByteDance/Seed-2.0-code":{"id":"deepinfra/ByteDance/Seed-2.0-code","name":"Seed 2.0 Code (Deep Infra)","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"deepinfra/ByteDance/Seed-2.0-mini":{"id":"deepinfra/ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini (Deep Infra)","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"deepinfra/openai/gpt-oss-20b":{"id":"deepinfra/openai/gpt-oss-20b","name":"GPT OSS 20B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.14}},"deepinfra/openai/gpt-oss-120b":{"id":"deepinfra/openai/gpt-oss-120b","name":"GPT OSS 120B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.037,"output":0.17}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75,"cache_read":0.35}},"groq/openai/gpt-oss-20b":{"id":"groq/openai/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/openai/gpt-oss-safeguard-20b":{"id":"groq/openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B (Groq)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/openai/gpt-oss-120b":{"id":"groq/openai/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest (Claude Opus 5.5)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5-5":{"id":"anthropic/claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":288000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":132000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"databricks/databricks-inkling":{"id":"databricks/databricks-inkling","name":"Inkling (Databricks)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.1,"cache_write":1.00002}},"databricks/databricks-gpt-oss-20b":{"id":"databricks/databricks-gpt-oss-20b","name":"GPT OSS 20B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"databricks/databricks-deepseek-v4-flash-0731":{"id":"databricks/databricks-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Databricks)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.014,"cache_write":0.14}},"databricks/databricks-gpt-oss-120b@eu":{"id":"databricks/databricks-gpt-oss-120b@eu","name":"GPT OSS 120B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"databricks/databricks-gpt-oss-120b":{"id":"databricks/databricks-gpt-oss-120b","name":"GPT OSS 120B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"databricks/databricks-deepseek-v4-pro-0813":{"id":"databricks/databricks-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Databricks)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.959999,"cache_read":0.132,"cache_write":1.31999}},"databricks/databricks-gpt-oss-20b@eu":{"id":"databricks/databricks-gpt-oss-20b@eu","name":"GPT OSS 20B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"together_ai/meta-models/Muse-Glimmer-30B":{"id":"together_ai/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Together AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"together_ai/thinkingmachines/Inkling":{"id":"together_ai/thinkingmachines/Inkling","name":"Inkling (Together AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Together AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Together AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together_ai/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"together_ai/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Together AI)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"together_ai/openai/gpt-oss-120b":{"id":"together_ai/openai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.1-codex-max":{"id":"azure/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"flexai/Step-3.7-Flash":{"id":"flexai/Step-3.7-Flash","name":"Step 3.7 Flash (FlexAI)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.03}},"flexai/DeepSeek-V4-Flash-0731":{"id":"flexai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (FlexAI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.009}},"flexai/Muse-Glimmer-30B":{"id":"flexai/Muse-Glimmer-30B","name":"Muse Glimmer 30B (FlexAI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.045}},"flexai/gpt-oss-20b":{"id":"flexai/gpt-oss-20b","name":"GPT OSS 20B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.02,"output":0.1,"cache_read":0.003}},"flexai/gpt-oss-120b":{"id":"flexai/gpt-oss-120b","name":"GPT OSS 120B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.17,"cache_read":0.0045}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["audio","image","text","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"google/deep-research-preview-04-2026":{"id":"google/deep-research-preview-04-2026","name":"Gemini Deep Research Preview","description":"Agentic model for autonomous multi-step research, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12}},"google/deep-research-max-preview-04-2026":{"id":"google/deep-research-max-preview-04-2026","name":"Deep Research Max Preview","description":"Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"xai/grok-latest":{"id":"xai/grok-latest","name":"Grok Latest (Grok 4.7)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"xai/grok-4.7":{"id":"xai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":3.2,"output":9.6,"cache_read":0.8,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":3.2,"output":9.6,"cache_read":0.8}}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"nebius/google/gemma-3-27b-it":{"id":"nebius/google/gemma-3-27b-it","name":"Gemma 3 27B IT (Nebius)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":110000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.1}},"nebius/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"nebius/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Nebius)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"nebius/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"nebius/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Nebius)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":979000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"nebius/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"nebius/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Nebius)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.3}},"nebius/nvidia/nemotron-3-super-120b-a12b":{"id":"nebius/nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B (Nebius)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.9,"cache_read":0.3}},"nebius/nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nebius/nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Nebius)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":1,"output":3,"cache_read":1}},"nebius/openai/gpt-oss-120b":{"id":"nebius/openai/gpt-oss-120b","name":"GPT OSS 120B (Nebius)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"scaleway/gemma-3-27b-it":{"id":"scaleway/gemma-3-27b-it","name":"Gemma 3 27B IT (Scaleway)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40000,"output":131072},"cost":{"input":0.287125,"output":0.57425}},"scaleway/deepseek-v4-flash-0731":{"id":"scaleway/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Scaleway)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":384000},"cost":{"input":0.45468,"output":0.90936,"cache_read":0.090936}},"scaleway/llama-3.3-70b-instruct":{"id":"scaleway/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct (Scaleway)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.02303,"output":1.02303}},"scaleway/gpt-oss-120b":{"id":"scaleway/gpt-oss-120b","name":"GPT OSS 120B (Scaleway)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.170505,"output":0.68202}},"ionos/meta-llama/Llama-3.3-70B-Instruct":{"id":"ionos/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (IONOS)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.738855,"output":0.738855}},"ionos/openai/gpt-oss-120b":{"id":"ionos/openai/gpt-oss-120b","name":"GPT OSS 120B (IONOS)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.170505,"output":0.738855}},"vertex/gemini-flash-latest":{"id":"vertex/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash, Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-2.5-flash-image":{"id":"vertex/gemini-2.5-flash-image","name":"Nano Banana (Vertex AI)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3.1-flash-lite@eu":{"id":"vertex/gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (Vertex AI, EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.5-flash-lite@eu":{"id":"vertex/gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.6-flash@eu":{"id":"vertex/gemini-3.6-flash@eu","name":"Gemini 3.6 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash":{"id":"vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash-lite":{"id":"vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.1-flash-image":{"id":"vertex/gemini-3.1-flash-image","name":"Nano Banana 2 (Vertex AI)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"vertex/gemini-3.1-pro-preview":{"id":"vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview (Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.5-flash":{"id":"vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3-pro-image":{"id":"vertex/gemini-3-pro-image","name":"Nano Banana Pro (Vertex AI)","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"vertex/gemini-3.7-flash@us":{"id":"vertex/gemini-3.7-flash@us","name":"Gemini 3.7 Flash (Vertex AI, US)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.7-flash@eu":{"id":"vertex/gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (Vertex AI, EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.1-flash-lite-image":{"id":"vertex/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite (Vertex AI)","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"vertex/gemini-3.8-flash@eu":{"id":"vertex/gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (Vertex AI, EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.7-flash":{"id":"vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash@us":{"id":"vertex/gemini-3.6-flash@us","name":"Gemini 3.6 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash-lite@us":{"id":"vertex/gemini-3.5-flash-lite@us","name":"Gemini 3.5 Flash Lite (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3-flash-preview":{"id":"vertex/gemini-3-flash-preview","name":"Gemini 3 Flash Preview (Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3.1-flash-lite@us":{"id":"vertex/gemini-3.1-flash-lite@us","name":"Gemini 3.1 Flash Lite (Vertex AI, US)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-pro-latest":{"id":"vertex/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview, Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.8-flash":{"id":"vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash@eu":{"id":"vertex/gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3.8-flash@us":{"id":"vertex/gemini-3.8-flash@us","name":"Gemini 3.8 Flash (Vertex AI, US)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash@us":{"id":"vertex/gemini-3.5-flash@us","name":"Gemini 3.5 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3.1-flash-lite":{"id":"vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"perplexityai/sonar-pro":{"id":"perplexityai/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"perplexityai/sonar-deep-research":{"id":"perplexityai/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"perplexityai/sonar":{"id":"perplexityai/sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":4096},"cost":{"input":1,"output":1}},"perplexityai/sonar-reasoning-pro":{"id":"perplexityai/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"cloudflare/@cf/meta/llama-guard-3-8b":{"id":"cloudflare/@cf/meta/llama-guard-3-8b","name":"Llama-Guard-3-8B (Cloudflare)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.484,"output":0.03}},"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma-SEA-LION-v4-27B-IT (Cloudflare)","description":"Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Cloudflare)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Cloudflare)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"cloudflare/@cf/zai-org/glm-4.7-flash":{"id":"cloudflare/@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash (Cloudflare)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5-Coder-32B-Instruct (Cloudflare)","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.66,"output":1}},"cloudflare/@cf/openai/gpt-oss-20b":{"id":"cloudflare/@cf/openai/gpt-oss-20b","name":"GPT OSS 20B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.3}},"cloudflare/@cf/openai/gpt-oss-120b":{"id":"cloudflare/@cf/openai/gpt-oss-120b","name":"GPT OSS 120B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.35,"output":0.75}},"minimax/MiniMax-M3":{"id":"minimax/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/MiniMax-M2.1":{"id":"minimax/MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M2.5":{"id":"minimax/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M2.7":{"id":"minimax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/MiniMax-M2":{"id":"minimax/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"tensorx/deepseek/deepseek-v4-flash-0731":{"id":"tensorx/deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (TensorX)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.0625}},"tensorx/deepseek/deepseek-v4.1-flash":{"id":"tensorx/deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (TensorX)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.5,"output":1.5,"cache_read":0.125}},"tensorx/deepseek/deepseek-v4-pro-0813":{"id":"tensorx/deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (TensorX)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":2,"output":4,"cache_read":0.5}},"tensorx/moonshotai/kimi-k2.5":{"id":"tensorx/moonshotai/kimi-k2.5","name":"Kimi K2.5 (TensorX)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125}},"amazon/google.gemma-3-12b-it":{"id":"amazon/google.gemma-3-12b-it","name":"Gemma 3 12B IT (Amazon Bedrock)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.09,"output":0.29}},"amazon/google.gemma-3-4b-it":{"id":"amazon/google.gemma-3-4b-it","name":"Gemma 3 4B IT (Amazon Bedrock)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.04,"output":0.08}},"amazon/openai.gpt-oss-safeguard-20b@us":{"id":"amazon/openai.gpt-oss-safeguard-20b@us","name":"GPT OSS Safeguard 20B (Amazon Bedrock, US)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.07,"output":0.2}},"amazon/openai.gpt-oss-safeguard-20b":{"id":"amazon/openai.gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B (Amazon Bedrock)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.07,"output":0.2}},"amazon/amazon.nova-lite-v1:0@us":{"id":"amazon/amazon.nova-lite-v1:0@us","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"amazon/mistral.voxtral-mini-3b-2507@us":{"id":"amazon/mistral.voxtral-mini-3b-2507@us","name":"Voxtral Mini 3B 2507 (Amazon Bedrock, US)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.04,"output":0.04}},"amazon/amazon.nova-micro-v1:0":{"id":"amazon/amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"amazon/amazon.nova-pro-v1:0":{"id":"amazon/amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"amazon/mistral.voxtral-small-24b-2507":{"id":"amazon/mistral.voxtral-small-24b-2507","name":"Voxtral Small 24B 2507 (Amazon Bedrock)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"amazon/amazon.nova-lite-v1:0":{"id":"amazon/amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"amazon/mistral.voxtral-small-24b-2507@us":{"id":"amazon/mistral.voxtral-small-24b-2507@us","name":"Voxtral Small 24B 2507 (Amazon Bedrock, US)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"amazon/google.gemma-3-12b-it@us":{"id":"amazon/google.gemma-3-12b-it@us","name":"Gemma 3 12B IT (Amazon Bedrock, US)","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.09,"output":0.29}},"amazon/google.gemma-3-27b-it":{"id":"amazon/google.gemma-3-27b-it","name":"Gemma 3 27B IT (Amazon Bedrock)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.23,"output":0.38}},"amazon/google.gemma-3-27b-it@us":{"id":"amazon/google.gemma-3-27b-it@us","name":"Gemma 3 27B IT (Amazon Bedrock, US)","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.23,"output":0.38}},"amazon/mistral.pixtral-large-2502-v1:0":{"id":"amazon/mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (Amazon Bedrock)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/amazon.nova-micro-v1:0@us":{"id":"amazon/amazon.nova-micro-v1:0@us","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"amazon/zai.glm-4.7-flash@us":{"id":"amazon/zai.glm-4.7-flash@us","name":"GLM-4.7-Flash (Amazon Bedrock, US)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":0.07,"output":0.4}},"amazon/moonshotai.kimi-k2.5":{"id":"amazon/moonshotai.kimi-k2.5","name":"Kimi K2.5 (Amazon Bedrock)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}},"amazon/amazon.nova-pro-v1:0@us":{"id":"amazon/amazon.nova-pro-v1:0@us","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"amazon/mistral.pixtral-large-2502-v1:0@us":{"id":"amazon/mistral.pixtral-large-2502-v1:0@us","name":"Pixtral Large (25.02) (Amazon Bedrock, US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/zai.glm-4.7-flash":{"id":"amazon/zai.glm-4.7-flash","name":"GLM-4.7-Flash (Amazon Bedrock)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":0.07,"output":0.4}},"amazon/moonshot.kimi-k2-thinking":{"id":"amazon/moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking (Amazon Bedrock)","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"cost":{"input":0.6,"output":2.5}},"amazon/google.gemma-3-4b-it@us":{"id":"amazon/google.gemma-3-4b-it@us","name":"Gemma 3 4B IT (Amazon Bedrock, US)","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":131072},"cost":{"input":0.04,"output":0.08}},"amazon/mistral.voxtral-mini-3b-2507":{"id":"amazon/mistral.voxtral-mini-3b-2507","name":"Voxtral Mini 3B 2507 (Amazon Bedrock)","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.04,"output":0.04}},"ovhcloud/gpt-oss-20b":{"id":"ovhcloud/gpt-oss-20b","name":"GPT OSS 20B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.18}},"ovhcloud/gpt-oss-120b":{"id":"ovhcloud/gpt-oss-120b","name":"GPT OSS 120B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.47}},"fireworks_ai/gpt-oss-120b":{"id":"fireworks_ai/gpt-oss-120b","name":"GPT OSS 120B (Fireworks AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.014}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Fireworks AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b":{"id":"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B (Fireworks AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Fireworks AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"fireworks_ai/accounts/fireworks/models/inkling":{"id":"fireworks_ai/accounts/fireworks/models/inkling","name":"Inkling (Fireworks AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-small-2603":{"id":"mistral/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"mistral/mistral-medium-2604":{"id":"mistral/mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwq-plus":{"id":"qwen/qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/deepseek-v4-flash-0731":{"id":"qwen/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Alibaba)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.022}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.23,"output":0.92}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4,"cache_read":0.32}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/deepseek-v4.1-flash":{"id":"qwen/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (Alibaba)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"qwen/qwen3-coder-next@eu":{"id":"qwen/qwen3-coder-next@eu","name":"Qwen3 Coder Next (EU)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25}},"qwen/qwen-vl-max":{"id":"qwen/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.16}},"qwen/qwen3-max@eu":{"id":"qwen/qwen3-max@eu","name":"Qwen3 Max (EU)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen-vl-plus":{"id":"qwen/qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63,"cache_read":0.042}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"qwen/deepseek-v4-pro-0813":{"id":"qwen/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Alibaba)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.066}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-mini-latest":{"id":"openai/gpt-mini-latest","name":"GPT Mini Latest (GPT-5.4 mini)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-pro-latest":{"id":"openai/gpt-pro-latest","name":"GPT Pro Latest (GPT-5.5 Pro)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"infomaniak/mistralai/Ministral-3-14B-Instruct-2512":{"id":"infomaniak/mistralai/Ministral-3-14B-Instruct-2512","name":"Ministral 3 14B (Infomaniak)","description":"Open vision-language model for efficient local deployment, instruction following, and tool use","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":262144},"cost":{"input":0.34101,"output":0.45468}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"stepfun":{"id":"stepfun","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/v1","name":"StepFun (China)","doc":"https://platform.stepfun.com/docs/zh/overview/concept","models":{"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-5-preview":{"id":"step-5-preview","name":"Step 5 Preview","description":"StepFun's next-generation flagship base model for coding and professional knowledge work, with native text, image, and video input and a 1M-token context window","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-09-16","last_updated":"2026-09-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.959,"output":2.741,"cache_read":0.048}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}}}},"hpc-ai":{"id":"hpc-ai","env":["HPC_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.hpc-ai.com/inference/v1","name":"HPC-AI","doc":"https://www.hpc-ai.com/doc/docs/quickstart/","models":{"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1002000,"output":128000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":202000},"cost":{"input":0.615,"output":2.46,"cache_read":0.133}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":195000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"v0":{"id":"v0","env":["V0_API_KEY"],"npm":"@ai-sdk/vercel","name":"v0","doc":"https://sdk.vercel.ai/providers/ai-sdk-providers/vercel","models":{"v0-1.5-lg":{"id":"v0-1.5-lg","name":"v0-1.5-lg","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":32000},"cost":{"input":15,"output":75}},"v0-1.5-md":{"id":"v0-1.5-md","name":"v0-1.5-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}},"v0-1.0-md":{"id":"v0-1.0-md","name":"v0-1.0-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}}}},"tencent-coding-plan":{"id":"tencent-coding-plan","env":["TENCENT_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/coding/v3","name":"Tencent Coding Plan (China)","doc":"https://cloud.tencent.com/document/product/1772/128947","models":{"hunyuan-2.0-thinking":{"id":"hunyuan-2.0-thinking","name":"Tencent HY 2.0 Think","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-2.0-instruct":{"id":"hunyuan-2.0-instruct","name":"Tencent HY 2.0 Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"tc-code-latest":{"id":"tc-code-latest","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-turbos":{"id":"hunyuan-turbos","name":"Hunyuan-TurboS","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-t1":{"id":"hunyuan-t1","name":"Hunyuan-T1","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"tempr":{"id":"tempr","env":["TEMPR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.temprhq.io/v1","name":"Tempr","doc":"https://temprhq.io/docs/gateway-reference.html","models":{"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-embedding-2":{"id":"google/gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1},"cost":{"input":0.2,"output":0}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral/zai-glm-5-2":{"id":"mistral/zai-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"mistral/mistral-embed":{"id":"mistral/mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":3072},"cost":{"input":0.1,"output":0}},"mistral/mistral-small-2603":{"id":"mistral/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"mistral/zai-glm-5-3":{"id":"mistral/zai-glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"mistral/mistral-medium-2604":{"id":"mistral/mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}}}},"inception":{"id":"inception","env":["INCEPTION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptionlabs.ai/v1/","name":"Inception","doc":"https://docs.inceptionlabs.ai/get-started/models","models":{"mercury-edit-2":{"id":"mercury-edit-2","name":"Mercury Edit 2","description":"Code editing dLLM for autocomplete (FIM) and next-edit suggestions","family":"mercury","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"mercury-2.5":{"id":"mercury-2.5","name":"Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11-01","release_date":"2026-09-08","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}}}},"modelis":{"id":"modelis","env":["MODELIS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://modelishub.com/v1","name":"Modelis","doc":"https://modelishub.com/pricing","models":{"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]},{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0983,"output":0.1966}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":3,"output":9}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.768,"output":3.072}}}},"opencode":{"id":"opencode","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/v1","name":"OpenCode Zen","doc":"https://opencode.ai/docs/zen","models":{"ling-3.0-flash-fin-free":{"id":"ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin Free","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"qwen3.6-plus-free":{"id":"qwen3.6-plus-free","name":"Qwen3.6 Plus Free","description":"Legacy model retained for compatibility with older integrations","family":"qwen-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"mimo-v2-pro-free":{"id":"mimo-v2-pro-free","name":"MiMo V2 Pro Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-pro-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"muse-spark-1.2-contributor-free":{"id":"muse-spark-1.2-contributor-free","name":"Muse Spark 1.2 Free","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"glm-5-free":{"id":"glm-5-free","name":"GLM-5 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"trinity-large-preview-free":{"id":"trinity-large-preview-free","name":"Trinity Large Preview","description":"Legacy model retained for compatibility with older integrations","family":"trinity","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-27","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0,"output":0}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"grok-4.7":{"id":"grok-4.7","name":"Grok 4.7 (30% Off)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.4,"output":4.2,"cache_read":0.35,"tiers":[{"input":2.8,"output":8.4,"cache_read":0.7,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.8,"output":8.4,"cache_read":0.7}}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"ling-3.0-tiny-free":{"id":"ling-3.0-tiny-free","name":"Ling-3.0-tiny Free","description":"Compact MoE model for responsive agents, instruction following, and multi-turn conversations","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"kimi-k2.5-free":{"id":"kimi-k2.5-free","name":"Kimi K2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"kimi-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-4.7-free":{"id":"glm-4.7-free","name":"GLM-4.7 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"laguna-s-2.1-free":{"id":"laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.45,"output":1.8}},"minimax-m3-free":{"id":"minimax-m3-free","name":"MiniMax-M3 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Legacy model retained for compatibility with older integrations","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"minimax-m2.5-free":{"id":"minimax-m2.5-free","name":"MiniMax-M2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"ring-2.6-1t-free":{"id":"ring-2.6-1t-free","name":"Ring 2.6 1T Free","description":"Legacy model retained for compatibility with older integrations","family":"ring-1t-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":66000},"status":"deprecated","cost":{"input":0,"output":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3,"cache_read":0.08}},"longcat-2.0-free":{"id":"longcat-2.0-free","name":"LongCat-2.0 Free","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-3-5-haiku":{"id":"claude-3-5-haiku","name":"Claude Haiku 3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"hy3-preview-free":{"id":"hy3-preview-free","name":"Hy3 preview Free","description":"Legacy model retained for compatibility with older integrations","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"space-bunny-free":{"id":"space-bunny-free","name":"Space Bunny Free","description":"Anonymous preview reasoning model for coding, agentic tasks, tool use, and multimodal input","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-23","last_updated":"2026-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":524288,"output":524288},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"nemotron-3-super-free":{"id":"nemotron-3-super-free","name":"Nemotron 3 Super Free","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"minimax-m2.1-free":{"id":"minimax-m2.1-free","name":"MiniMax-M2.1 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.6-flash-free":{"id":"mimo-v2.6-flash-free","name":"MiMo-V2.6-Flash Free","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"gemini-3-pro":{"id":"gemini-3-pro","name":"Gemini 3 Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"nemotron-3-ultra-free":{"id":"nemotron-3-ultra-free","name":"Nemotron 3 Ultra Free","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"mimo-v2.5-free":{"id":"mimo-v2.5-free","name":"MiMo V2.5 Free","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"grok-code":{"id":"grok-code","name":"Grok Code Fast 1","description":"Legacy model retained for compatibility with older integrations","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-20","last_updated":"2025-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"ling-2.6-flash-free":{"id":"ling-2.6-flash-free","name":"Ling 2.6 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"ling-flash-free","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":32800},"status":"deprecated","cost":{"input":0,"output":0}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"deepseek-v4-flash-free":{"id":"deepseek-v4-flash-free","name":"DeepSeek V4 Flash Free","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"hy3-free":{"id":"hy3-free","name":"Hy3 Free","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":190000,"input":192000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"nemotron-3.5-lightning-free":{"id":"nemotron-3.5-lightning-free","name":"Nemotron 3.5 Lightning Free","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"mimo-v2-omni-free":{"id":"mimo-v2-omni-free","name":"MiMo V2 Omni Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-omni-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1,"output":2,"cache_read":0.2}},"ling-3.0-flash-free":{"id":"ling-3.0-flash-free","name":"Ling-3.0-flash Free","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"north-mini-code-free":{"id":"north-mini-code-free","name":"North Mini Code Free","description":"Cohere coding model for practical software engineering and agentic edits","family":"north-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.84,"cache_read":0.145}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"big-pickle":{"id":"big-pickle","name":"Big Pickle","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"big-pickle","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":160000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"mimo-v2-flash-free":{"id":"mimo-v2-flash-free","name":"MiMo V2 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-flash-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.1}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"x-preview-f-free":{"id":"x-preview-f-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"muse-spark-1.3-contributor-free":{"id":"muse-spark-1.3-contributor-free","name":"Muse Spark 1.3 Free","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}}}},"kenari":{"id":"kenari","env":["KENARI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://kenari.id/v1","name":"Kenari","doc":"https://kenari.id/docs","models":{"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"gemini-2-5-flash":{"id":"gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"nemotron-3-nano-30b-a3b":{"id":"nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"gemini-2-5-flash-lite":{"id":"gemini-2-5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"glm-4-7-flash:free":{"id":"glm-4-7-flash:free","name":"GLM-4.7-Flash (Free)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":272000,"output":16384},"cost":{"input":0,"output":0}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"step-3-7-flash:free":{"id":"step-3-7-flash:free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"kimi-k2-6:free":{"id":"kimi-k2-6:free","name":"Kimi K2.6 (Free)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"mistral-large:free":{"id":"mistral-large:free","name":"Mistral Large (Free)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"mimo-v2-5:free":{"id":"mimo-v2-5:free","name":"MiMo-V2.5 (Free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"hy3:free":{"id":"hy3:free","name":"Hy3 (Free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"mistral-medium-3-5:free":{"id":"mistral-medium-3-5:free","name":"Mistral Medium 3.5 (Free)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-3":{"id":"glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-1":{"id":"glm-5-1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b:free":{"id":"nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super 120B A12B (Free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"grok-imagine-image-2-0":{"id":"grok-imagine-image-2-0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":8000,"output":0},"cost":{"input":0,"output":0}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash (Free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0,"output":0}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-3-1-flash-tts":{"id":"gemini-3-1-flash-tts","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0,"output":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"kimi-k2-7-code:free":{"id":"kimi-k2-7-code:free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}}}},"kimi-code-plan-global":{"id":"kimi-code-plan-global","env":["KIMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kimi.ai/coding/v1","name":"Kimi For Coding (kimi.ai)","doc":"https://www.kimi.ai/code/docs/en/kimi-code/models.html","models":{"kimi-for-coding-highspeed":{"id":"kimi-for-coding-highspeed","name":"Kimi For Coding HighSpeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-for-coding":{"id":"kimi-for-coding","name":"kimi-for-coding","description":"Kimi coding model with more efficient thinking and up to 1M context, available through Kimi Code","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3-256k":{"id":"k3-256k","name":"Kimi K3-256K","description":"256K-context version of Kimi K3, reducing token consumption for shorter coding sessions","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3":{"id":"k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"trustedrouter":{"id":"trustedrouter","env":["TRUSTEDROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.trustedrouter.com/v1","name":"TrustedRouter","doc":"https://trustedrouter.com/docs","models":{"trustedrouter/cheap":{"id":"trustedrouter/cheap","name":"Cheap","description":"TrustedRouter low-cost routing alias that prefers inexpensive healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth-code":{"id":"trustedrouter/synth-code","name":"Synth Code","description":"TrustedRouter code synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/e2e":{"id":"trustedrouter/e2e","name":"End-to-End Encrypted","description":"TrustedRouter privacy routing alias for end-to-end encrypted provider routes where available.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/zdr":{"id":"trustedrouter/zdr","name":"Zero Data Retention","description":"TrustedRouter privacy routing alias that prefers zero data retention model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/fast":{"id":"trustedrouter/fast","name":"Fast","description":"TrustedRouter speed routing alias that prefers low-latency healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth":{"id":"trustedrouter/synth","name":"Synth","description":"TrustedRouter synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/auto":{"id":"trustedrouter/auto","name":"Auto","description":"TrustedRouter automatic routing alias that chooses a healthy supported model endpoint for the request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}}}},"wafer.ai":{"id":"wafer.ai","env":["WAFER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://pass.wafer.ai/v1","name":"Wafer","doc":"https://docs.wafer.ai/wafer-pass","models":{"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.14,"output":4.8,"cache_read":0.19,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.07,"cache_write":0,"tiers":[{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0}}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"General Language Model 5.1 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.1,"cache_write":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.1,"cache_read":0.2,"cache_write":0}},"glm5.2-fast":{"id":"glm5.2-fast","name":"GLM5.2-Fast","description":"The same model served for high TPS.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":10.25,"cache_read":0.5,"cache_write":0}}}},"zhipuai":{"id":"zhipuai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/paas/v4","name":"Zhipu AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":5,"output":22,"cache_read":1.2,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.3-flashx":{"id":"glm-5.3-flashx","name":"GLM-5.3-FlashX","description":"High-speed GLM-5.3-Flash serving option for coding and agent workflows","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-18","last_updated":"2026-09-18","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.37,"output":1.25,"cache_read":0.075,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.6v-flash":{"id":"glm-4.6v-flash","name":"GLM-4.6V-Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}}}},"lynkr":{"id":"lynkr","env":["LYNKR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:8081/v1","name":"Lynkr","doc":"https://github.com/Fast-Editor/Lynkr","models":{"lynkr-auto":{"id":"lynkr-auto","name":"Lynkr Auto (complexity routing)","description":"Virtual model: Lynkr scores each request on complexity and routes it to the tier model the user configured (local Ollama/llama.cpp for simple requests, configured cloud providers for complex ones).","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}}}},"meganova":{"id":"meganova","env":["MEGANOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.meganova.ai/v1","name":"Meganova","doc":"https://docs.meganova.ai","models":{"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.3}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo V2 Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3}},"Qwen/Qwen3.5-Plus":{"id":"Qwen/Qwen3.5-Plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.6}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.2,"output":0.6}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.26,"output":0.38}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":0.88}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.4}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.28,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.02,"output":0.04}},"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.8}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.6}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.2,"output":0.8}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.8,"output":2.56}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.45,"output":1.9}}}},"ovhcloud":{"id":"ovhcloud","env":["OVHCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://oai.endpoints.kepler.ai.cloud.ovh.net/v1","name":"OVHcloud AI Endpoints","doc":"https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//","models":{"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.47,"output":3.19}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"Qwen3Guard-Gen-8B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.11,"output":0.11}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder-30B-A3B-Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.26}},"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"Qwen3Guard-Gen-0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.18}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.18}},"meta-llama-3_3-70b-instruct":{"id":"meta-llama-3_3-70b-instruct","name":"Meta-Llama-3_3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.74,"output":0.74}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-18","last_updated":"2026-05-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":4.25}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen2.5-VL-72B-Instruct","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":1.01,"output":1.01}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.47,"output":3.19}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral-Small-3.2-24B-Instruct-2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-16","last_updated":"2025-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.31}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.09,"output":0.47}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral-Nemo-Instruct-2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.14,"output":0.14}}}},"requesty":{"id":"requesty","env":["REQUESTY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://router.requesty.ai/v1","name":"Requesty","doc":"https://requesty.ai/solution/llm-routing/models","models":{"ring-2.6-1t":{"id":"ring-2.6-1t","name":"ring-2.6-1t","description":"Inclusion AI ring-2.6-1t","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"nemotron-3-ultra-nvfp4":{"id":"nemotron-3-ultra-nvfp4","name":"nemotron-3-ultra-nvfp4","description":"Nemotron-3-Ultra-550B-A55B-NVFP4 is a frontier-scale large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities. It is optimized for the most demanding workloads, including complex multi-step agents, long-context analysis, and high-accuracy reasoning over code, math, and science. The model employs a hybrid Latent Mixture-of-Experts (LatentMoE) architecture, utilizing interleaved Mamba-2 and MoE layers, along with select Attention layers. Like the Super model, the Ultra model incorporates Multi-Token Prediction (MTP) layers for faster text generation and improved quality, and it is trained using an NVFP4 pre-training recipe to maximize compute efficiency. The model has 55B active parameters and 550B parameters in total.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":30}},"claude-opus-5@eu":{"id":"claude-opus-5@eu","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"ling-2.6-1t":{"id":"ling-2.6-1t","name":"ling-2.6-1t","description":"Inclusion AI ling-2.6-1t","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"gpt-4o-mini@eu":{"id":"gpt-4o-mini@eu","name":"GPT-4o mini (EU)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.165,"output":0.66,"cache_read":0.0825}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"nemotron-3.5-lightning-30b-a3b":{"id":"nemotron-3.5-lightning-30b-a3b","name":"nemotron-3.5-lightning-30b-a3b","description":"NVIDIA Nemotron 3.5 Lightning 30B-A3B is a hybrid Mamba-2 + MoE + Attention model with 30B total and 3B active parameters, pre-trained on over 20T tokens with an NVFP4 recipe and Multi-Token Prediction for fast generation. Up to 1M token context for long-running autonomous agents, sub-agent workhorse deployments, and agentic workflows. Supports reasoning and tool calling. English and coding languages plus Spanish, French, German, Italian, and Japanese. Open weights under the OpenMDW License Agreement v1.1. Part of the NVIDIA Nemotron family.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"mistral-medium-3-5@eu":{"id":"mistral-medium-3-5@eu","name":"mistral-medium-3-5@eu","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5@eu":{"id":"gpt-5@eu","name":"GPT-5 (EU)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"deepseek-v4-flash-0731@eu":{"id":"deepseek-v4-flash-0731@eu","name":"DeepSeek V4 Flash 0731 (EU)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"seed-2.0-pro":{"id":"seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"kimi-k3@eu":{"id":"kimi-k3@eu","name":"Kimi K3 (EU)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"seed-1.8":{"id":"seed-1.8","name":"seed-1.8","description":"Optimized specifically for multimodal agent scenarios. It features enhanced agent capabilities, upgraded multimodal comprehension, and more flexible context management.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.6}},"mistral-medium-latest@eu":{"id":"mistral-medium-latest@eu","name":"Mistral Medium (latest) (EU)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"glm-5.2-fast","description":"GLM-5.2 introduces a robust 1M-token context and advanced, multi-effort coding capabilities to significantly enhance performance on long-horizon tasks. Its new IndexShare architecture and improved MTP layer simultaneously boost efficiency by reducing per-token FLOPs and increasing speculative decoding lengths. A 743B-parameter model in Zhipu AI's GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-13","last_updated":"2026-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"deepseek-v4-pro-0813@eu":{"id":"deepseek-v4-pro-0813@eu","name":"DeepSeek V4 Pro 0813 (EU)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"inkling-256k":{"id":"inkling-256k","name":"inkling-256k","description":"Inkling 256K is the extended context variant of Inkling, a large MoE hybrid reasoning model from Thinking Machines with audio and vision input support and a 256K context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"claude-opus-4-8@eu":{"id":"claude-opus-4-8@eu","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"claude-fable-5@eu":{"id":"claude-fable-5@eu","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"nemotron-3-nano-omni@eu":{"id":"nemotron-3-nano-omni@eu","name":"nemotron-3-nano-omni@eu","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"gpt-4.1@eu":{"id":"gpt-4.1@eu","name":"GPT-4.1 (EU)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.2,"output":8.8,"cache_read":0.55}},"gemini-3.1-flash-lite@eu":{"id":"gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.275,"output":1.65,"cache_read":0.0275,"cache_write":0.091663}},"nvidia-nemotron-3-super-120b-a12b":{"id":"nvidia-nemotron-3-super-120b-a12b","name":"nvidia-nemotron-3-super-120b-a12b","description":"NVIDIA Nemotron 3 Super is a hybrid Mixture-of-Experts (MoE) model engineered for highest compute efficiency and accuracy in multi-agent applications and specialized agentic systems. It is optimized to run many collaborating agents per application on a single GPU, delivering high accuracy for reasoning, tool use, and instruction following.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.5}},"gemini-3.5-flash-lite@eu":{"id":"gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.033}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.7-code@eu":{"id":"kimi-k2.7-code@eu","name":"Kimi K2.7 Code (EU)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.31}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"glm-5.1@eu":{"id":"glm-5.1@eu","name":"GLM-5.1 (EU)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-haiku-4-5@eu":{"id":"claude-haiku-4-5@eu","name":"Claude Haiku 4.5 (latest) (EU)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"claude-sonnet-4-5@eu":{"id":"claude-sonnet-4-5@eu","name":"Claude Sonnet 4.5 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125,"tiers":[{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25}}},"laguna-xs.2":{"id":"laguna-xs.2","name":"Laguna XS.2","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"leanstral-1-5@eu":{"id":"leanstral-1-5@eu","name":"leanstral-1-5@eu","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-6@eu":{"id":"claude-sonnet-4-6@eu","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":7,"cache_read":0.15}},"gpt-5.5@eu":{"id":"gpt-5.5@eu","name":"GPT-5.5 (EU)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"ling-3.0-tiny":{"id":"ling-3.0-tiny","name":"ling-3.0-tiny","description":"Ling-3.0-tiny is an efficient 7.9B parameter MoE model from inclusionAI with only 1.3B active parameters per token. Built for responsive agents, reliable instruction following and multi turn conversation, with a 256K context window, native function calling, prompt caching and switchable Thinking and Instant modes.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"claude-opus-5-5@eu":{"id":"claude-opus-5-5@eu","name":"Claude Opus 5.5 (EU)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"gpt-5-mini@eu":{"id":"gpt-5-mini@eu","name":"GPT-5 Mini (EU)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.275,"output":2.2,"cache_read":0.0275}},"mistral-medium-3-5":{"id":"mistral-medium-3-5","name":"mistral-medium-3-5","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.34,"cache_read":0.07}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gpt-5.4@eu":{"id":"gpt-5.4@eu","name":"GPT-5.4 (EU)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"grok-4.2-beta":{"id":"grok-4.2-beta","name":"grok-4.2-beta","description":"Grok 4.20 Beta is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently precise and truthful responses.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":2,"output":6,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.4,"cache_write":4}}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"nemotron-lightning-3.5-30b-a3b":{"id":"nemotron-lightning-3.5-30b-a3b","name":"nemotron-lightning-3.5-30b-a3b","description":"Nemotron-Lightning-3.5-30B-A3B is a 30B-parameter Mixture-of-Experts language model (3B active) from NVIDIA's Nemotron-H family, built on a hybrid Mamba-Transformer architecture for efficient long-context inference. Like other models in the family, it responds to queries by first generating a reasoning trace and then concluding with a final response, with reasoning behavior configurable through a flag in the chat template. It includes a multi-token prediction (MTP) speculative decoding head for low-latency serving.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-15","last_updated":"2026-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"gpt-5-nano@eu":{"id":"gpt-5-nano@eu","name":"GPT-5 Nano (EU)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.055,"output":0.44,"cache_read":0.0055}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":9,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":9}}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"seed-2.0-mini":{"id":"seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6@eu":{"id":"kimi-k2.6@eu","name":"Kimi K2.6 (EU)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"qwen3.5-2b","description":"Qwen3.5-2B is a compact yet capable model from Alibaba's Qwen3.5 series. It features a 262K token context window, support for 201 languages, thinking/reasoning mode, and tool calling for agentic workflows. A strong choice for prototyping, fine-tuning, and efficient multilingual deployments.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.02,"output":0.1}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":1.2}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.583}},"ling-2.6-flash":{"id":"ling-2.6-flash","name":"ling-2.6-flash","description":"Inclusion AI ling-2.6-flash","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"gemini-3.7-flash@eu":{"id":"gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"gpt-4.1-nano@eu":{"id":"gpt-4.1-nano@eu","name":"GPT-4.1 nano (EU)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.11,"output":0.44,"cache_read":0.0275}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"gpt-5.1@eu":{"id":"gpt-5.1@eu","name":"GPT-5.1 (EU)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"gemini-2.5-flash-lite@eu":{"id":"gemini-2.5-flash-lite@eu","name":"Gemini 2.5 Flash-Lite (EU)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.18333}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"nemotron-3-nano-omni","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"claude-sonnet-4@eu":{"id":"claude-sonnet-4@eu","name":"Claude Sonnet 4 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"deepseek-v4.1-flash@eu":{"id":"deepseek-v4.1-flash@eu","name":"DeepSeek V4.1 Flash (EU)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"qwen3.8-2.4T-A95B@eu":{"id":"qwen3.8-2.4T-A95B@eu","name":"Qwen3.8 2.4T A95B (EU)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"qwen3.8-flash-next@eu":{"id":"qwen3.8-flash-next@eu","name":"Qwen3.8 Flash Next (EU)","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":33,"cache_read":0.55}},"claude-sonnet-5@eu":{"id":"claude-sonnet-5@eu","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"gemini-3.8-flash@eu":{"id":"gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"glm-5.3@eu":{"id":"glm-5.3@eu","name":"GLM-5.3 (EU)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gemini-2.5-flash@eu":{"id":"gemini-2.5-flash@eu","name":"Gemini 2.5 Flash (EU)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.55}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"glm-5.2@eu":{"id":"glm-5.2@eu","name":"GLM-5.2 (EU)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gpt-5.6-sol@eu":{"id":"gpt-5.6-sol@eu","name":"GPT-5.6 Sol (EU)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44}},"seed-2.0-code":{"id":"seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"devstral-latest@eu":{"id":"devstral-latest@eu","name":"devstral-latest@eu","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gemini-2.5-pro@eu":{"id":"gemini-2.5-pro@eu","name":"Gemini 2.5 Pro (EU)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":2.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75}}},"gpt-5.6-terra@eu":{"id":"gpt-5.6-terra@eu","name":"GPT-5.6 Terra (EU)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.1}},"glm-5.3-flash@eu":{"id":"glm-5.3-flash@eu","name":"GLM-5.3-Flash (EU)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":1.2}},"deepseek-v4-pro@eu":{"id":"deepseek-v4-pro@eu","name":"DeepSeek V4 Pro (EU)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"devstral-latest":{"id":"devstral-latest","name":"devstral-latest","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"o4-mini@eu":{"id":"o4-mini@eu","name":"o4-mini (EU)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.21,"output":4.84,"cache_read":0.3025}},"mistral-small-2603@eu":{"id":"mistral-small-2603@eu","name":"Mistral Small 4 (EU)","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"qwen3.8-2.4T-A95B":{"id":"qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gemini-3.5-flash@eu":{"id":"gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.65,"output":9.9,"cache_read":0.165,"cache_write":1.7413}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02}}},"nemotron-3.5-content-safety":{"id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"nvidia-nemotron-3-ultra":{"id":"nvidia-nemotron-3-ultra","name":"nvidia-nemotron-3-ultra","description":"NVIDIA Nemotron 3 Ultra is NVIDIA's strongest open-weights reasoning model, positioned near GPT-5.4 Mini (xhigh) and ahead of DeepSeek V4-Flash and Qwen3.5-397B-A17B.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":2.5}},"claude-fable-5.1@eu":{"id":"claude-fable-5.1@eu","name":"Claude Fable 5.1 (EU)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"gpt-6-luna@eu":{"id":"gpt-6-luna@eu","name":"GPT-6 Luna (EU)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.12,"output":0.6,"cache_read":0.012,"tiers":[{"input":0.24,"output":0.9,"cache_read":0.024,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.24,"output":0.9,"cache_read":0.024}}},"claude-opus-4-6@eu":{"id":"claude-opus-4-6@eu","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"leanstral-1-5":{"id":"leanstral-1-5","name":"leanstral-1-5","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"gpt-5.6-luna@eu":{"id":"gpt-5.6-luna@eu","name":"GPT-5.6 Luna (EU)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022}},"claude-opus-4-7@eu":{"id":"claude-opus-4-7@eu","name":"Claude Opus 4.7 (EU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"kat-coder-pro":{"id":"kat-coder-pro","name":"kat-coder-pro","description":"KAT-Coder-Pro V2 by KwaiKAT is a non-reasoning model optimized for agentic coding. It delivers strong performance on reasoning-style tasks while requiring significantly fewer output tokens than peer models. With the 1210 release, it achieved a score of 64 on the Artificial Analysis Intelligence Index, placing it in the global Top 10 and ranking first among all non-reasoning models.","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":1.2}},"laguna-m.1":{"id":"laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"gpt-6-sol@eu":{"id":"gpt-6-sol@eu","name":"GPT-6 Sol (EU)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.4,"output":12,"cache_read":0.24,"tiers":[{"input":4.8,"output":18,"cache_read":0.48,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.8,"output":18,"cache_read":0.48}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"minimax-m3@eu":{"id":"minimax-m3@eu","name":"MiniMax-M3 (EU)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.032,"cache_write":0.4}},"gpt-4.1-mini@eu":{"id":"gpt-4.1-mini@eu","name":"GPT-4.1 mini (EU)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.44,"output":1.76,"cache_read":0.11}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"claude-opus-4-5@eu":{"id":"claude-opus-4-5@eu","name":"Claude Opus 4.5 (latest) (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"tiers":[{"input":4,"output":15,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}}}},"mistral":{"id":"mistral","env":["MISTRAL_API_KEY"],"npm":"@ai-sdk/mistral","name":"Mistral","doc":"https://docs.mistral.ai/getting-started/models/","models":{"open-mistral-nemo":{"id":"open-mistral-nemo","name":"Open Mistral Nemo","description":"Legacy model retained for compatibility with older integrations","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"codestral-latest":{"id":"codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"voxtral-mini-tts-latest":{"id":"voxtral-mini-tts-latest","name":"Voxtral Mini TTS (latest)","description":"Multilingual text-to-speech model with zero-shot voice cloning","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"mistral-medium-2508":{"id":"mistral-medium-2508","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral-small-latest":{"id":"mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"zai-glm-5-2":{"id":"zai-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"ministral-8b-latest":{"id":"ministral-8b-latest","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"devstral-medium-latest":{"id":"devstral-medium-latest","name":"Devstral 2 (latest)","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"open-mixtral-8x22b":{"id":"open-mixtral-8x22b","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":2,"output":6}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"magistral-medium-latest":{"id":"magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"pixtral-12b":{"id":"pixtral-12b","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"mistral-embed":{"id":"mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":3072},"cost":{"input":0.1,"output":0}},"devstral-small-2505":{"id":"devstral-small-2505","name":"Devstral Small 2505","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"devstral-medium-2507":{"id":"devstral-medium-2507","name":"Devstral Medium","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"magistral-small":{"id":"magistral-small","name":"Magistral Small","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.5,"output":1.5}},"voxtral-mini-latest":{"id":"voxtral-mini-latest","name":"Voxtral Mini (latest)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"labs-devstral-small-2512":{"id":"labs-devstral-small-2512","name":"Devstral Small 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"devstral-latest":{"id":"devstral-latest","name":"Devstral 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"devstral-small-2507":{"id":"devstral-small-2507","name":"Devstral Small","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"pixtral-large-latest":{"id":"pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}},"voxtral-small-latest":{"id":"voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"open-mixtral-8x7b":{"id":"open-mixtral-8x7b","name":"Mixtral 8x7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-01","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.7,"output":0.7}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"zai-glm-5-3":{"id":"zai-glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"mistral-medium-2604":{"id":"mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"open-mistral-7b":{"id":"open-mistral-7b","name":"Mistral 7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":0.25,"output":0.25}},"ministral-3b-latest":{"id":"ministral-3b-latest","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}}}},"amazon-bedrock":{"id":"amazon-bedrock","env":["AWS_ACCESS_KEY_ID","AWS_SECRET_ACCESS_KEY","AWS_REGION","AWS_BEARER_TOKEN_BEDROCK"],"npm":"@ai-sdk/amazon-bedrock","name":"Amazon Bedrock","doc":"https://docs.aws.amazon.com/bedrock/latest/userguide/models-supported.html","models":{"google.gemma-3-12b-it":{"id":"google.gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open multimodal Gemma instruction model for multilingual text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.09,"output":0.29}},"google.gemma-3-4b-it":{"id":"google.gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open multimodal Gemma instruction model for efficient text generation and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.04,"output":0.08}},"eu.anthropic.claude-fable-5":{"id":"eu.anthropic.claude-fable-5","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"qwen.qwen3-coder-480b-a35b-v1:0":{"id":"qwen.qwen3-coder-480b-a35b-v1:0","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.45,"output":1.8}},"google.gemma-4-31b":{"id":"google.gemma-4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.14,"output":0.4}},"us.writer.palmyra-x5-v1:0":{"id":"us.writer.palmyra-x5-v1:0","name":"Palmyra X5 (US)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"input":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"google.gemma-4-e2b":{"id":"google.gemma-4-e2b","name":"Gemma 4 E2B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.04,"output":0.08}},"us.anthropic.claude-opus-4-7":{"id":"us.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (US)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"global.openai.gpt-6-astra":{"id":"global.openai.gpt-6-astra","name":"GPT-6 Astra (Global)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"eu.amazon.nova-lite-v1:0":{"id":"eu.amazon.nova-lite-v1:0","name":"Nova Lite (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.069,"output":0.276,"cache_read":0.01725,"cache_write":0.069}},"global.anthropic.claude-opus-5-5":{"id":"global.anthropic.claude-opus-5-5","name":"Claude Opus 5.5 (Global)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"deepseek.r1-v1:0":{"id":"deepseek.r1-v1:0","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"openai.gpt-oss-safeguard-20b":{"id":"openai.gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.2}},"anthropic.claude-opus-5-5":{"id":"anthropic.claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"global.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"global.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (Global)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"eu.amazon.nova-2-lite-v1:0":{"id":"eu.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (EU)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.374,"output":3.157,"cache_read":0.0935,"cache_write":0.374}},"us.amazon.nova-pro-v1:0":{"id":"us.amazon.nova-pro-v1:0","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"jp.anthropic.claude-opus-5-5":{"id":"jp.anthropic.claude-opus-5-5","name":"Claude Opus 5.5 (JP)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.22,"cache_write":5.5}},"openai.gpt-5.6-luna":{"id":"openai.gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"us.meta.llama4-maverick-17b-instruct-v1:0":{"id":"us.meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct (US)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.24,"output":0.97}},"apac.amazon.nova-micro-v1:0":{"id":"apac.amazon.nova-micro-v1:0","name":"Nova Micro (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.037,"output":0.148,"cache_read":0.00925,"cache_write":0.037}},"eu.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"eu.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"eu.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"eu.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"global.openai.gpt-6-sol":{"id":"global.openai.gpt-6-sol","name":"GPT-6 Sol (Global)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"qwen.qwen3-coder-30b-a3b-v1:0":{"id":"qwen.qwen3-coder-30b-a3b-v1:0","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}},"us.amazon.nova-premier-v1:0":{"id":"us.amazon.nova-premier-v1:0","name":"Nova Premier (US)","description":"Multimodal model for complex analysis, long-context understanding, tool use, and model distillation","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":10000},"status":"deprecated","cost":{"input":2.5,"output":12.5,"cache_read":0.625,"cache_write":2.5}},"anthropic.claude-sonnet-4-6":{"id":"anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"meta.llama3-1-70b-instruct-v1:0":{"id":"meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"us.writer.palmyra-x4-v1:0":{"id":"us.writer.palmyra-x4-v1:0","name":"Palmyra X4 (US)","description":"Enterprise language model for workflow automation, coding, data analysis, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-10-09","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"apac.amazon.nova-pro-v1:0":{"id":"apac.amazon.nova-pro-v1:0","name":"Nova Pro (APAC)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.84,"output":3.36,"cache_read":0.21,"cache_write":0.84}},"global.anthropic.claude-fable-5-1":{"id":"global.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (Global)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"us.anthropic.claude-sonnet-5":{"id":"us.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (US)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"amazon.nova-micro-v1:0":{"id":"amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"minimax.minimax-m2.5":{"id":"minimax.minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":98304},"cost":{"input":0.3,"output":1.2}},"xai.grok-4.3":{"id":"xai.grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-06-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"us.amazon.nova-lite-v1:0":{"id":"us.amazon.nova-lite-v1:0","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"amazon.nova-pro-v1:0":{"id":"amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.8,"output":3.2,"cache_read":0.2,"cache_write":0.8}},"qwen.qwen3-32b-v1:0":{"id":"qwen.qwen3-32b-v1:0","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.15,"output":0.6}},"openai.gpt-5.5":{"id":"openai.gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":5.5,"output":33,"cache_read":0.55}},"mistral.voxtral-small-24b-2507":{"id":"mistral.voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.1,"output":0.3}},"global.amazon.nova-2-lite-v1:0":{"id":"global.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (Global)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.3}},"us.openai.gpt-5.6-terra":{"id":"us.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (US)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"mistral.devstral-2-123b":{"id":"mistral.devstral-2-123b","name":"Devstral 2 123B","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.4,"output":2}},"us.xai.grok-4.6":{"id":"us.xai.grok-4.6","name":"Grok 4.6 (US)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"jp.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"jp.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (JP)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"anthropic.claude-fable-5":{"id":"anthropic.claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"jp.anthropic.claude-opus-4-8":{"id":"jp.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (JP)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.openai.gpt-5.6-luna":{"id":"us.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (US)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"us.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"us.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (US)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"au.anthropic.claude-opus-5":{"id":"au.anthropic.claude-opus-5","name":"Claude Opus 5 (AU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"nvidia.nemotron-nano-12b-v2":{"id":"nvidia.nemotron-nano-12b-v2","name":"NVIDIA Nemotron Nano 12B v2 VL BF16","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.2,"output":0.6}},"us.amazon.nova-2-lite-v1:0":{"id":"us.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (US)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.0825,"cache_write":0.33}},"us.openai.gpt-6-astra":{"id":"us.openai.gpt-6-astra","name":"GPT-6 Astra (US)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"openai.gpt-5.6-terra":{"id":"openai.gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"anthropic.claude-opus-4-6-v1":{"id":"anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.anthropic.claude-opus-4-1-20250805-v1:0":{"id":"us.anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"us.meta.llama3-1-8b-instruct-v1:0":{"id":"us.meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct (US)","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"eu.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"eu.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (EU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"global.anthropic.claude-opus-4-7":{"id":"global.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (Global)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"apac.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"apac.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (APAC)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"deepseek.v3.2":{"id":"deepseek.v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.62,"output":1.85}},"qwen.qwen3-235b-a22b-2507-v1:0":{"id":"qwen.qwen3-235b-a22b-2507-v1:0","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.22,"output":0.88}},"amazon.nova-lite-v1:0":{"id":"amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.06,"output":0.24,"cache_read":0.015,"cache_write":0.06}},"global.anthropic.claude-opus-5":{"id":"global.anthropic.claude-opus-5","name":"Claude Opus 5 (Global)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai.gpt-oss-20b-1:0":{"id":"openai.gpt-oss-20b-1:0","name":"gpt-oss-20b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":128000},"cost":{"input":0.07,"output":0.3}},"anthropic.claude-opus-5":{"id":"anthropic.claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"meta.llama3-3-70b-instruct-v1:0":{"id":"meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"minimax.minimax-m2":{"id":"minimax.minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204608,"output":128000},"cost":{"input":0.3,"output":1.2}},"mistral.mistral-large-3-675b-instruct":{"id":"mistral.mistral-large-3-675b-instruct","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.5,"output":1.5}},"in.openai.gpt-5.6-luna":{"id":"in.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (India)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"eu.anthropic.claude-opus-4-8":{"id":"eu.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"global.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"global.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (Global)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"zai.glm-4.7":{"id":"zai.glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2}},"au.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"au.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (AU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"eu.anthropic.claude-opus-5-5":{"id":"eu.anthropic.claude-opus-5-5","name":"Claude Opus 5.5 (EU)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.22,"cache_write":5.5}},"global.anthropic.claude-sonnet-4-6":{"id":"global.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Global)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"jp.anthropic.claude-sonnet-4-6":{"id":"jp.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (JP)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"google.gemma-3-27b-it":{"id":"google.gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Largest open Gemma 3 instruction model for multilingual text generation and visual understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.38}},"eu.anthropic.claude-opus-4-6-v1":{"id":"eu.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.anthropic.claude-opus-5-5":{"id":"us.anthropic.claude-opus-5-5","name":"Claude Opus 5.5 (US)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.22,"cache_write":5.5}},"us.anthropic.claude-opus-5":{"id":"us.anthropic.claude-opus-5","name":"Claude Opus 5 (US)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.openai.gpt-5.6-sol":{"id":"us.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (US)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"us.moonshotai.kimi-k3":{"id":"us.moonshotai.kimi-k3","name":"Kimi K3 (US)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"anthropic.claude-sonnet-5":{"id":"anthropic.claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"mistral.magistral-small-2509":{"id":"mistral.magistral-small-2509","name":"Magistral Small 1.2","description":"Open multimodal reasoning model for transparent analysis of text and images","family":"magistral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":40000},"cost":{"input":0.5,"output":1.5}},"mistral.pixtral-large-2502-v1:0":{"id":"mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"anthropic.claude-opus-4-5-20251101-v1:0":{"id":"anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"nvidia.nemotron-nano-9b-v2":{"id":"nvidia.nemotron-nano-9b-v2","name":"NVIDIA Nemotron Nano 9B v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.06,"output":0.23}},"eu.amazon.nova-micro-v1:0":{"id":"eu.amazon.nova-micro-v1:0","name":"Nova Micro (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.04,"output":0.16,"cache_read":0.01,"cache_write":0.04}},"zai.glm-5":{"id":"zai.glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2}},"us.meta.llama3-1-70b-instruct-v1:0":{"id":"us.meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct (US)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"jp.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"jp.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (JP)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"nvidia.nemotron-nano-3-30b":{"id":"nvidia.nemotron-nano-3-30b","name":"NVIDIA Nemotron Nano 3 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.06,"output":0.24}},"mistral.ministral-3-3b-instruct":{"id":"mistral.ministral-3-3b-instruct","name":"Ministral 3 3B","description":"Compact open vision-language model for edge deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1,"output":0.1}},"us.openai.gpt-6-luna":{"id":"us.openai.gpt-6-luna","name":"GPT-6 Luna (US)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.11,"output":0.55,"cache_read":0.011,"cache_write":0.1375,"tiers":[{"input":0.22,"output":0.825,"cache_read":0.022,"cache_write":0.275,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.22,"output":0.825,"cache_read":0.022,"cache_write":0.275}}},"global.openai.gpt-5.6-sol":{"id":"global.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (Global)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"google.gemma-4-26b-a4b":{"id":"google.gemma-4-26b-a4b","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.13,"output":0.4}},"xai.grok-4.6":{"id":"xai.grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"global.xai.grok-4.6":{"id":"global.xai.grok-4.6","name":"Grok 4.6 (Global)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"au.anthropic.claude-opus-4-6-v1":{"id":"au.anthropic.claude-opus-4-6-v1","name":"AU Anthropic Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen.qwen3-vl-235b-a22b":{"id":"qwen.qwen3-vl-235b-a22b","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262000},"cost":{"input":0.53,"output":2.66}},"anthropic.claude-fable-5-1":{"id":"anthropic.claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"jp.anthropic.claude-sonnet-5":{"id":"jp.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (JP)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"qwen.qwen3-coder-next":{"id":"qwen.qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.5,"output":1.2}},"anthropic.claude-opus-4-8":{"id":"anthropic.claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"au.anthropic.claude-sonnet-5":{"id":"au.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (AU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"meta.llama4-scout-17b-instruct-v1:0":{"id":"meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":10000000,"output":8192},"cost":{"input":0.17,"output":0.66}},"openai.gpt-oss-safeguard-120b":{"id":"openai.gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"us.meta.llama3-3-70b-instruct-v1:0":{"id":"us.meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct (US)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"global.openai.gpt-5.6-luna":{"id":"global.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (Global)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"au.anthropic.claude-sonnet-4-6":{"id":"au.anthropic.claude-sonnet-4-6","name":"AU Anthropic Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"openai.gpt-oss-120b":{"id":"openai.gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.15,"output":0.6}},"deepseek.v3-v1:0":{"id":"deepseek.v3-v1:0","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"moonshotai.kimi-k2.5":{"id":"moonshotai.kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.6,"output":3}},"eu.anthropic.claude-sonnet-4-6":{"id":"eu.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"us-gov.openai.gpt-oss-120b-1:0":{"id":"us-gov.openai.gpt-oss-120b-1:0","name":"gpt-oss-120b (GovCloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.18,"output":0.72}},"us.anthropic.claude-fable-5":{"id":"us.anthropic.claude-fable-5","name":"Claude Fable 5 (US)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"openai.gpt-6-sol":{"id":"openai.gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":16.5,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":16.5,"cache_read":0.44,"cache_write":5.5}}},"us.deepseek.r1-v1:0":{"id":"us.deepseek.r1-v1:0","name":"DeepSeek-R1 (US)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"eu.anthropic.claude-opus-5":{"id":"eu.anthropic.claude-opus-5","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"ca.amazon.nova-lite-v1:0":{"id":"ca.amazon.nova-lite-v1:0","name":"Nova Lite (CA)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.064,"output":0.256,"cache_read":0.016,"cache_write":0.064}},"us.openai.gpt-6-sol":{"id":"us.openai.gpt-6-sol","name":"GPT-6 Sol (US)","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":16.5,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":16.5,"cache_read":0.44,"cache_write":5.5}}},"au.anthropic.claude-opus-4-7":{"id":"au.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (AU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.meta.llama4-scout-17b-instruct-v1:0":{"id":"us.meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct (US)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":10000000,"output":8192},"cost":{"input":0.17,"output":0.66}},"global.anthropic.claude-sonnet-5":{"id":"global.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (Global)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"global.anthropic.claude-fable-5":{"id":"global.anthropic.claude-fable-5","name":"Claude Fable 5 (Global)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"apac.amazon.nova-lite-v1:0":{"id":"apac.amazon.nova-lite-v1:0","name":"Nova Lite (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.063,"output":0.252,"cache_read":0.01575,"cache_write":0.063}},"us.anthropic.claude-opus-4-8":{"id":"us.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (US)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"nvidia.nemotron-super-3-120b":{"id":"nvidia.nemotron-super-3-120b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.65}},"global.openai.gpt-6-luna":{"id":"global.openai.gpt-6-luna","name":"GPT-6 Luna (Global)","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"us-gov.openai.gpt-oss-20b-1:0":{"id":"us-gov.openai.gpt-oss-20b-1:0","name":"gpt-oss-20b (GovCloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.084,"output":0.36}},"us.anthropic.claude-opus-4-6-v1":{"id":"us.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (US)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"meta.llama3-1-8b-instruct-v1:0":{"id":"meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"writer.palmyra-x5-v1:0":{"id":"writer.palmyra-x5-v1:0","name":"Palmyra X5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"input":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"anthropic.claude-opus-4-1-20250805-v1:0":{"id":"anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"us.anthropic.claude-sonnet-4-6":{"id":"us.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (US)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"global.anthropic.claude-opus-4-6-v1":{"id":"global.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (Global)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"eu.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"jp.anthropic.claude-opus-4-7":{"id":"jp.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (JP)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"au.anthropic.claude-opus-5-5":{"id":"au.anthropic.claude-opus-5-5","name":"Claude Opus 5.5 (AU)","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.22,"cache_write":5.5}},"openai.gpt-6-luna":{"id":"openai.gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.11,"output":0.55,"cache_read":0.011,"cache_write":0.1375,"tiers":[{"input":0.22,"output":0.825,"cache_read":0.022,"cache_write":0.275,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.22,"output":0.825,"cache_read":0.022,"cache_write":0.275}}},"writer.palmyra-x4-v1:0":{"id":"writer.palmyra-x4-v1:0","name":"Palmyra X4","description":"Enterprise language model for workflow automation, coding, data analysis, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-10-09","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"global.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"global.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (Global)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic.claude-opus-4-7":{"id":"anthropic.claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"us.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (US)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"eu.anthropic.claude-sonnet-5":{"id":"eu.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"zai.glm-4.7-flash":{"id":"zai.glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.07,"output":0.4}},"moonshot.kimi-k2-thinking":{"id":"moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16000},"cost":{"input":0.6,"output":2.5}},"eu.amazon.nova-pro-v1:0":{"id":"eu.amazon.nova-pro-v1:0","name":"Nova Pro (EU)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.92,"output":3.68,"cache_read":0.23,"cache_write":0.92}},"global.openai.gpt-5.6-terra":{"id":"global.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (Global)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai.gpt-6-astra":{"id":"openai.gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"us.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"us.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (US)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"meta.llama4-maverick-17b-instruct-v1:0":{"id":"meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.24,"output":0.97}},"qwen.qwen3-next-80b-a3b":{"id":"qwen.qwen3-next-80b-a3b","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262000},"cost":{"input":0.15,"output":1.2}},"eu.mistral.pixtral-large-2502-v1:0":{"id":"eu.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (EU)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"au.anthropic.claude-opus-4-8":{"id":"au.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (AU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.anthropic.claude-fable-5-1":{"id":"us.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (US)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"global.moonshotai.kimi-k3":{"id":"global.moonshotai.kimi-k3","name":"Kimi K3 (Global)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"in.openai.gpt-5.6-terra":{"id":"in.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (India)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"openai.gpt-oss-20b":{"id":"openai.gpt-oss-20b","name":"gpt-oss-20b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.07,"output":0.3}},"mistral.ministral-3-14b-instruct":{"id":"mistral.ministral-3-14b-instruct","name":"Ministral 14B 3.0","description":"Open vision-language model for efficient local deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"au.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"au.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (AU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"openai.gpt-5.6-sol":{"id":"openai.gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"minimax.minimax-m2.1":{"id":"minimax.minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2}},"us.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"us.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"eu.anthropic.claude-opus-4-7":{"id":"eu.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (EU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.amazon.nova-2-lite-v1:0":{"id":"jp.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (JP)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.396,"output":3.311,"cache_read":0.099,"cache_write":0.396}},"us.mistral.pixtral-large-2502-v1:0":{"id":"us.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"jp.anthropic.claude-opus-5":{"id":"jp.anthropic.claude-opus-5","name":"Claude Opus 5 (JP)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"global.anthropic.claude-sonnet-4-20250514-v1:0":{"id":"global.anthropic.claude-sonnet-4-20250514-v1:0","name":"Claude Sonnet 4 (Global)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"global.anthropic.claude-opus-4-8":{"id":"global.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (Global)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai.gpt-5.4":{"id":"openai.gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"mistral.voxtral-mini-3b-2507":{"id":"mistral.voxtral-mini-3b-2507","name":"Voxtral Mini 3B 2507","description":"Open audio-language model for speech transcription, audio understanding, and voice-driven tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":4096},"cost":{"input":0.04,"output":0.04}},"openai.gpt-oss-120b-1:0":{"id":"openai.gpt-oss-120b-1:0","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":128000},"cost":{"input":0.15,"output":0.6}},"amazon.nova-2-lite-v1:0":{"id":"amazon.nova-2-lite-v1:0","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.0825,"cache_write":0.33}},"us.amazon.nova-micro-v1:0":{"id":"us.amazon.nova-micro-v1:0","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875,"cache_write":0.035}},"mistral.ministral-3-8b-instruct":{"id":"mistral.ministral-3-8b-instruct","name":"Ministral 3 8B","description":"Compact open vision-language model for edge deployment, instruction following, and tool use","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.15}}}},"synthetic":{"id":"synthetic","env":["SYNTHETIC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.synthetic.new/openai/v1","name":"Synthetic","doc":"https://synthetic.new/pricing","models":{"hf:moonshotai/Kimi-K3":{"id":"hf:moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.45}},"hf:moonshotai/Kimi-K2.7-Code":{"id":"hf:moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4":{"id":"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.3}},"hf:zai-org/GLM-5.2":{"id":"hf:zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"hf:zai-org/GLM-5.3-Flash":{"id":"hf:zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}},"hf:zai-org/GLM-4.7-Flash":{"id":"hf:zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.1,"output":0.5,"cache_read":0.1}},"hf:deepseek-ai/DeepSeek-V4.1-Flash":{"id":"hf:deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.6,"output":1.2,"cache_read":0.03}},"hf:Qwen/Qwen3.6-27B":{"id":"hf:Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.6,"cache_read":0.45}},"hf:openai/gpt-oss-120b":{"id":"hf:openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1}},"hf:MiniMaxAI/MiniMax-M3":{"id":"hf:MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.6,"output":1.2,"cache_read":0.6}}}},"llmgateway":{"id":"llmgateway","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"DevPass (LLM Gateway)","doc":"https://llmgateway.io/docs","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"mimo-v2.6-pro":{"id":"mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"seed-1-6-flash-250715":{"id":"seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":3.125}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.08,"output":0.35,"cache_read":0.05}},"gpt-4o-transcribe":{"id":"gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":2.5,"output":10}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"nemotron-3.5-lightning":{"id":"nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.08,"output":0.2}},"qwen3-vl-235b-a22b-thinking":{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.98,"output":3.95}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1010000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"qwen-coder-plus":{"id":"qwen-coder-plus","name":"Qwen Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8,"cache_read":0.04,"cache_write":0.25}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.088,"output":0.25,"cache_read":0.025}},"fugu-max":{"id":"fugu-max","name":"Fugu Max","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.15}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct (2507)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.09,"output":0.58}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"glm-4.5-x":{"id":"glm-4.5-x","name":"GLM-4.5 X","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"grok-4":{"id":"grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"llama-4-maverick-17b-instruct":{"id":"llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":2048},"cost":{"input":0.27,"output":0.85}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":0.72,"output":2.3,"cache_read":0.144,"cache_write":0}},"fugu-ultra-v2.0":{"id":"fugu-ultra-v2.0","name":"Fugu Ultra v2.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-11","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":228700,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.36,"output":0.87,"reasoning":8.4}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4,"cache_read":0.08,"cache_write":0.5}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"minimax-m2.1-lightning":{"id":"minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38,"cache_read":0.6,"cache_write":3.75}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"glm-4.5-airx":{"id":"glm-4.5-airx","name":"GLM-4.5 AirX","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"llama-4-scout-17b-instruct":{"id":"llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":2048},"cost":{"input":0.18,"output":0.59}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"qwen3-235b-a22b-fp8":{"id":"qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":8192},"cost":{"input":0.2,"output":0.8}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"glm-4-32b-0414-128k":{"id":"glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.1}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0.07,"output":0.27}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"ling-3.0-flash":{"id":"ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"qwen3-vl-30b-a3b-instruct":{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"seed-1-6-250915":{"id":"seed-1-6-250915","name":"Seed 1.6 (250915)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"llama-3-70b-instruct":{"id":"llama-3-70b-instruct","name":"Llama 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"qwen3-vl-flash":{"id":"qwen3-vl-flash","name":"Qwen3 VL Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking (2507)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.3,"output":3}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":1050000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"minimax-text-01":{"id":"minimax-text-01","name":"MiniMax Text 01","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-4.6v-flashx":{"id":"glm-4.6v-flashx","name":"GLM-4.6V FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"grok-4-20-beta-0309-non-reasoning":{"id":"grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.135,"output":0.4}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.08,"output":0.32,"cache_read":0.017,"cache_write":0.375}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"llama-3.2-11b-instruct":{"id":"llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.07,"output":0.33}},"inkling-small":{"id":"inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01,"cache_write":0}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"mimo-v2.6-flash":{"id":"mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"seed-1-8-251228":{"id":"seed-1-8-251228","name":"Seed 1.8 (251228)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.38,"output":1.98,"cache_read":0.19,"cache_write":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"seed-1-6-250615":{"id":"seed-1-6-250615","name":"Seed 1.6 (250615)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen35-397b-a17b":{"id":"qwen35-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16,"cache_write":0}},"ernie-4.5-vl-424b-a47b":{"id":"ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":123000},"cost":{"input":0.42,"output":1.25}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.132,"output":0.528,"cache_read":0.033}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.57,"output":2.3,"cache_read":0.5}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.2,"output":0.8}},"hy-mt2-plus":{"id":"hy-mt2-plus","name":"Hy-MT2 Plus","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"Hy","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.074,"output":0.295}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.931,"output":2.93,"cache_read":0.173,"cache_write":0}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"grok-4-20-beta-0309-reasoning":{"id":"grok-4-20-beta-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gemini-pro-latest":{"id":"gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"minimax-m2.5-highspeed":{"id":"minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.2}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-4o-mini-transcribe":{"id":"gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":1.25,"output":5}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"llama-3.2-3b-instruct":{"id":"llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4-7":{"id":"grok-4-7","name":"Grok 4.7","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"atria-dawn-preview":{"id":"atria-dawn-preview","name":"Atria Dawn Preview","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-12","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025,"cache_write":0}},"qwen-plus-latest":{"id":"qwen-plus-latest","name":"Qwen Plus Latest","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":8192},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.1,"output":0.1}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.27,"output":1.1}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"auto":{"id":"auto","name":"Auto Route","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"llama-3.1-70b-instruct":{"id":"llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"status":"beta","cost":{"input":0.72,"output":0.72}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"custom":{"id":"custom","name":"Custom Model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"codestral-2508":{"id":"codestral-2508","name":"Codestral","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":0.9}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}}}},"sap-ai-core":{"id":"sap-ai-core","env":["AICORE_SERVICE_KEY"],"npm":"@jerome-benoit/sap-ai-provider-v2","name":"SAP AI Core","doc":"https://help.sap.com/docs/sap-ai-core","models":{"anthropic--claude-4-opus":{"id":"anthropic--claude-4-opus","name":"anthropic--claude-4-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"nvidia--llama-3.2-nv-embedqa-1b":{"id":"nvidia--llama-3.2-nv-embedqa-1b","name":"nvidia--llama-3.2-nv-embedqa-1b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.07,"output":0}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"sap-abap-1":{"id":"sap-abap-1","name":"sap-abap-1","description":"SAP-hosted model for ABAP code generation and enterprise development tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.48,"output":1.7}},"anthropic--claude-4.8-opus":{"id":"anthropic--claude-4.8-opus","name":"anthropic--claude-4.8-opus","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"anthropic--claude-3-haiku":{"id":"anthropic--claude-3-haiku","name":"anthropic--claude-3-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.08,"output":0.26}},"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"anthropic--claude-4.5-haiku":{"id":"anthropic--claude-4.5-haiku","name":"anthropic--claude-4.5-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic--claude-4.7-opus":{"id":"anthropic--claude-4.7-opus","name":"anthropic--claude-4.7-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-embedding":{"id":"gemini-embedding","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":3072}},"amazon--nova-pro":{"id":"amazon--nova-pro","name":"amazon--nova-pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.56,"output":2.13}},"amazon--nova-lite":{"id":"amazon--nova-lite","name":"amazon--nova-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.3,"output":2.37}},"sonar-pro":{"id":"sonar-pro","name":"sonar-pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"anthropic--claude-3.5-sonnet":{"id":"anthropic--claude-3.5-sonnet","name":"anthropic--claude-3.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.09,"output":0}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"amazon--nova-micro":{"id":"amazon--nova-micro","name":"amazon--nova-micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.1}},"anthropic--claude-4.5-opus":{"id":"anthropic--claude-4.5-opus","name":"anthropic--claude-4.5-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic--claude-4-sonnet":{"id":"anthropic--claude-4-sonnet","name":"anthropic--claude-4-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"gemini-3.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-25","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"anthropic--claude-3-sonnet":{"id":"anthropic--claude-3-sonnet","name":"anthropic--claude-3-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-04","last_updated":"2024-03-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"sonar-deep-research":{"id":"sonar-deep-research","name":"sonar-deep-research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"mistralai--mistral-small":{"id":"mistralai--mistral-small","name":"mistralai--mistral-small","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.28}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-17","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"anthropic--claude-3.7-sonnet":{"id":"anthropic--claude-3.7-sonnet","name":"anthropic--claude-3.7-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"gpt-5.6-luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":9.44,"cache_read":0.12}},"gpt-5.5":{"id":"gpt-5.5","name":"gpt-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"anthropic--claude-4.6-opus":{"id":"anthropic--claude-4.6-opus","name":"anthropic--claude-4.6-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistralai--mistral-medium-instruct":{"id":"mistralai--mistral-medium-instruct","name":"mistralai--mistral-medium-instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.36,"output":1.22}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.32}},"anthropic--claude-3-opus":{"id":"anthropic--claude-3-opus","name":"anthropic--claude-3-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic--claude-4.6-sonnet":{"id":"anthropic--claude-4.6-sonnet","name":"anthropic--claude-4.6-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"sonar":{"id":"sonar","name":"sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"amazon--titan-embed-text":{"id":"amazon--titan-embed-text","name":"amazon--titan-embed-text","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-04-30","last_updated":"2024-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.14,"output":0}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"gpt-5.6-terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"mistralai--mistral-medium":{"id":"mistralai--mistral-medium","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"anthropic--claude-4.5-sonnet":{"id":"anthropic--claude-4.5-sonnet","name":"anthropic--claude-4.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"gpt-5.6-sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"cohere--command-a-reasoning":{"id":"cohere--command-a-reasoning","name":"cohere--command-a-reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.63,"output":5.05}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"gemini-3.1-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}}}},"vivgrid":{"id":"vivgrid","env":["VIVGRID_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.vivgrid.com/v1","name":"Vivgrid","doc":"https://docs.vivgrid.com/models","models":{"viv-fast":{"id":"viv-fast","name":"Viv Fast","description":"Fast coding model","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-09","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":256000},"cost":{"input":0.13,"output":0.4,"cache_read":0.05}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.31,"output":1.23,"cache_read":0.01}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-opus-5-5":{"id":"claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.5,"cache_write":12.5}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1.25,"cache_write":12.5}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.35,"output":3,"reasoning":3,"cache_read":0.05}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.2,"cache_read":0.3}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.15}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.28,"output":0.42}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.3,"reasoning":0.3,"cache_read":0.03}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}}}},"klokintegration":{"id":"klokintegration","env":["KLOKINTEGRATION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-gw.klok.ipaas.se/proxy/kloker-key/v1","name":"klokintegration.se","doc":"https://klokintegration.se/docs/ai-api","models":{"Kloker-Integration-Developer":{"id":"Kloker-Integration-Developer","name":"Kloker Integration Developer","description":"Knows the customer integration environment and Klok best practices. Opinionated about implementation. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection. Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker":{"id":"Kloker","name":"Kloker","description":"Cheap general model with a clean context. Nothing from the customer environment is packed in. It tracks the current best open source model. The Klok team verifies it and upgrades it periodically.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker-Integration-Architect":{"id":"Kloker-Integration-Architect","name":"Kloker Integration Architect","description":"Knows the customer integration environment and Klok best practices. Opinionated about structure. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection (data contracts, CloudEvents, event-driven flows). Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}}}},"google-vertex":{"id":"google-vertex","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex","name":"Vertex","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/models","models":{"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-2.5-flash-tts":{"id":"gemini-2.5-flash-tts","name":"Gemini 2.5 Flash TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-5-5@default":{"id":"claude-opus-5-5@default","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60,"cache_read":0.05}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":120,"cache_read":0.2}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"gemini-2.5-pro-tts":{"id":"gemini-2.5-pro-tts","name":"Gemini 2.5 Pro TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":1,"output":20}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"meta/llama-4-maverick-17b-128e-instruct-maas":{"id":"meta/llama-4-maverick-17b-128e-instruct-maas","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":8192},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.35,"output":1.15}},"meta/llama-3.3-70b-instruct-maas":{"id":"meta/llama-3.3-70b-instruct-maas","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.72,"output":0.72}},"xai/grok-4.20-reasoning":{"id":"xai/grok-4.20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-non-reasoning":{"id":"xai/grok-4.20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":30000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast","description":"Fast Grok model for responsive chat, tool-assisted work, and low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":500000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"deepseek-ai/deepseek-v3.1-maas":{"id":"deepseek-ai/deepseek-v3.1-maas","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":1.7,"cache_read":0.06}},"deepseek-ai/deepseek-v3.2-maas":{"id":"deepseek-ai/deepseek-v3.2-maas","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-17","last_updated":"2026-04-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"moonshotai/kimi-k2-thinking-maas":{"id":"moonshotai/kimi-k2-thinking-maas","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"zai-org/glm-5-maas":{"id":"zai-org/glm-5-maas","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"zai-org/glm-4.7-maas":{"id":"zai-org/glm-4.7-maas","name":"GLM-4.7","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-06","last_updated":"2026-01-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.2,"cache_read":0.06}},"zai-org/glm-5.2-maas":{"id":"zai-org/glm-5.2-maas","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"qwen/qwen3-235b-a22b-instruct-2507-maas":{"id":"qwen/qwen3-235b-a22b-instruct-2507-maas","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.22,"output":0.88}},"openai/gpt-oss-20b-maas":{"id":"openai/gpt-oss-20b-maas","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.07,"output":0.25,"cache_read":0.007}},"openai/gpt-oss-120b-maas":{"id":"openai/gpt-oss-120b-maas","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.09,"output":0.36}}}},"evroc":{"id":"evroc","env":["EVROC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.think.evroc.com/v1","name":"evroc","doc":"https://docs.evroc.com/products/think/overview.html","models":{"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.144,"output":0.575}},"intfloat/multilingual-e5-large-instruct":{"id":"intfloat/multilingual-e5-large-instruct","name":"E5 Multi-Lingual Large Embeddings 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"cost":{"input":0.114,"output":0.114}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.87,"output":3.5}},"Qwen/Qwen3-Reranker-4B":{"id":"Qwen/Qwen3-Reranker-4B","name":"Qwen3 Reranker 4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.0575,"output":0}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.345,"output":1.38}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":4096},"cost":{"input":0.115,"output":0.115}},"mistralai/Mistral-Medium-3.5-128B":{"id":"mistralai/Mistral-Medium-3.5-128B","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.725,"output":6.9}},"mistralai/Voxtral-Small-24B-2507":{"id":"mistralai/Voxtral-Small-24B-2507","name":"Voxtral Small 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["audio","text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.4375,"output":5.75}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1.4375,"output":5.75}},"nvidia/Llama-3.3-70B-Instruct-FP8":{"id":"nvidia/Llama-3.3-70B-Instruct-FP8","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.15,"output":1.15}},"KBLab/kb-whisper-large":{"id":"KBLab/kb-whisper-large","name":"KB Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"evroc/roc":{"id":"evroc/roc","name":"roc","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":2.875,"output":11.516}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper 3 Large","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/whisper-large-v3-turbo":{"id":"openai/whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.23,"output":0.92}}}},"tokengo":{"id":"tokengo","env":["TOKENGO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokengo.com/v1","name":"TokenGo","doc":"https://www.tokengo.com/docs","models":{"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.19,"output":0.71,"cache_read":0.06}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.2174,"output":0.326,"cache_read":0.06}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.098,"output":0.196,"cache_read":0.028}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.025,"cache_read":0.015}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.89,"output":3.2647,"cache_read":0.2226}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":2.65,"cache_read":0.2}}}},"submodel":{"id":"submodel","env":["SUBMODEL_INSTAGEN_ACCESS_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.submodel.ai/v1","name":"submodel","doc":"https://submodel.gitbook.io","models":{"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.3}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.5,"output":2.15}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"zai-org/GLM-4.5-FP8":{"id":"zai-org/GLM-4.5-FP8","name":"GLM 4.5 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.8}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}}}},"kosmik":{"id":"kosmik","env":["KOSMIK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.koscompute.com/v1","name":"Kosmik Compute","doc":"https://api.koscompute.com/docs/","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.35,"output":2.2,"cache_read":0.09}}}},"tencent-token-plan":{"id":"tencent-token-plan","env":["TENCENT_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/plan/v3","name":"Tencent Token Plan","doc":"https://cloud.tencent.com/document/product/1823/130060","models":{"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}}}},"togetherai":{"id":"togetherai","env":["TOGETHER_API_KEY"],"npm":"@ai-sdk/togetherai","name":"Together AI","doc":"https://docs.together.ai/docs/serverless-models","models":{"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":1.04,"output":1.04}},"meta-llama/Meta-Llama-3-8B-Instruct-Lite":{"id":"meta-llama/Meta-Llama-3-8B-Instruct-Lite","name":"Meta Llama 3 8B Instruct Lite","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"pearl-ai/gemma-4-31b-it":{"id":"pearl-ai/gemma-4-31b-it","name":"Pearl AI Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.28,"output":0.86}},"deepcogito/cogito-v2-1-671b":{"id":"deepcogito/cogito-v2-1-671b","name":"Cogito v2.1 671B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"cogito","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":1.25,"output":1.25}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max","xhigh","high","medium","low","none"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"essentialai/Rnj-1-Instruct":{"id":"essentialai/Rnj-1-Instruct","name":"Rnj-1 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"rnj","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"google/gemma-3n-E4B-it":{"id":"google/gemma-3n-E4B-it","name":"Gemma 3N E4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.06,"output":0.12}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.39,"output":0.97}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":500000},"cost":{"input":1.25,"output":3.75,"cache_read":0.125}},"Qwen/Qwen2.5-7B-Instruct-Turbo":{"id":"Qwen/Qwen2.5-7B-Instruct-Turbo","name":"Qwen 2.5 7B Instruct Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3}},"Qwen/Qwen3-Coder-Next-FP8":{"id":"Qwen/Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-02-03","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":1.2}},"Qwen/Qwen3-235B-A22B-Instruct-2507-tput":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507-tput","name":"Qwen3 235B A22B Instruct 2507 FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":130000},"status":"deprecated","cost":{"input":0.6,"output":3.6,"cache_read":0.35}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":2,"output":2}},"Qwen/Qwen3.6-Plus":{"id":"Qwen/Qwen3.6-Plus","name":"Qwen3.6 Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":500000},"cost":{"input":0.5,"output":3}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2-24B-A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163839,"output":163839},"status":"deprecated","cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek-ai/DeepSeek-V3-1":{"id":"deepseek-ai/DeepSeek-V3-1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":1.7}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":1.25,"output":1.25}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":250000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131000},"cost":{"input":1.2,"output":4.5,"cache_read":0.2}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":2.8}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-07","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":164000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1,"output":3.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":400000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512300,"output":512300},"cost":{"input":0.6,"output":3.6,"cache_read":0.2}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6}}}},"helicone":{"id":"helicone","env":["HELICONE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai-gateway.helicone.ai/v1","name":"Helicone","doc":"https://helicone.ai/models","models":{"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"xAI Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"grok-3":{"id":"grok-3","name":"xAI Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"llama-4-scout":{"id":"llama-4-scout","name":"Meta Llama 4 Scout 17B 16E","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.08,"output":0.3}},"llama-guard-4":{"id":"llama-guard-4","name":"Meta Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":1024},"cost":{"input":0.21,"output":0.21}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral-Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":6}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"xAI Grok 4.1 Fast Non-Reasoning","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":1.5}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":20,"output":40}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1 Codex","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Google Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.19999999999999998}},"gpt-4o":{"id":"gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-4.1-mini-2025-04-14":{"id":"gpt-4.1-mini-2025-04-14","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"glm-4.6":{"id":"glm-4.6","name":"Zai GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.44999999999999996,"output":1.5}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Anthropic: Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":0.7999999999999999,"output":4,"cache_read":0.08,"cache_write":1}},"o1-mini":{"id":"o1-mini","name":"OpenAI: o1-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"grok-4":{"id":"grok-4","name":"xAI Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-09","last_updated":"2024-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"gpt-5-codex":{"id":"gpt-5-codex","name":"OpenAI: GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Meta Llama 4 Maverick 17B 128E","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.95}},"gpt-5-mini":{"id":"gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"o4-mini":{"id":"o4-mini","name":"OpenAI o4 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"o3-mini":{"id":"o3-mini","name":"OpenAI o3 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2023-10","release_date":"2023-10-01","last_updated":"2023-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"kimi-k2-0711":{"id":"kimi-k2-0711","name":"Kimi K2 (07/11)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.5700000000000001,"output":2.3}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"OpenAI GPT-4.1 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"xAI: Grok 4 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":40960},"cost":{"input":0.29,"output":0.59}},"gpt-5-nano":{"id":"gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.049999999999999996,"output":0.39999999999999997,"cache_read":0.005}},"claude-3.5-sonnet-v2":{"id":"claude-3.5-sonnet-v2","name":"Anthropic: Claude 3.5 Sonnet v2","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"OpenAI GPT-5.1 Chat","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Anthropic: Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"deepseek-tng-r1t2-chimera":{"id":"deepseek-tng-r1t2-chimera","name":"DeepSeek TNG R1T2 Chimera","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-02","last_updated":"2025-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":163840},"cost":{"input":0.3,"output":1.2}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.03,"output":0.13}},"o1":{"id":"o1","name":"OpenAI: o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"mistral-small":{"id":"mistral-small","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.2}},"chatgpt-4o-latest":{"id":"chatgpt-4o-latest","name":"OpenAI ChatGPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":5,"output":20,"cache_read":2.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.09999999999999999,"output":0.3}},"sonar-pro":{"id":"sonar-pro","name":"Perplexity Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":3,"output":15}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":262144},"cost":{"input":0.48,"output":2}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"xAI Grok 4 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Anthropic: Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5-pro":{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":15,"output":120}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"OpenAI GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-09","release_date":"2024-09-30","last_updated":"2024-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"deepseek-v3.1-terminus":{"id":"deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.27,"output":1,"cache_read":0.21600000000000003}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Anthropic: Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5.1":{"id":"gpt-5.1","name":"OpenAI GPT-5.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Meta Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.59,"output":0.7899999999999999}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Meta Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":0.13,"output":0.39}},"grok-3-mini":{"id":"grok-3-mini","name":"xAI Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"gemma-3-12b-it":{"id":"gemma-3-12b-it","name":"Google Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.049999999999999996,"output":0.09999999999999999}},"qwen2.5-coder-7b-fast":{"id":"qwen2.5-coder-7b-fast","name":"Qwen2.5 Coder 7B fast","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-15","last_updated":"2024-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.03,"output":0.09}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"OpenAI GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":16384},"cost":{"input":0.14,"output":1.4}},"o3-pro":{"id":"o3-pro","name":"OpenAI o3 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"OpenAI GPT-OSS 20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.049999999999999996,"output":0.19999999999999998}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Google Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.3125,"cache_write":1.25}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Google Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.3}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Anthropic: Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"gemma2-9b-it":{"id":"gemma2-9b-it","name":"Google Gemma 2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-25","last_updated":"2024-06-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.01,"output":0.03}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Anthropic: Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"hermes-2-pro-llama-3-8b":{"id":"hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-27","last_updated":"2024-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.14,"output":0.14}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"qwen3-235b-a22b-thinking":{"id":"qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":81920},"cost":{"input":0.3,"output":2.9000000000000004}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"xAI Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-25","last_updated":"2024-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000},"cost":{"input":0.19999999999999998,"output":1.5,"cache_read":0.02}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Anthropic: Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gpt-4.1":{"id":"gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":41000,"output":41000},"cost":{"input":0.08,"output":0.29}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi K2 (09/05)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.5,"output":2,"cache_read":0.39999999999999997}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Meta Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.049999999999999996}},"llama-3.1-8b-instruct-turbo":{"id":"llama-3.1-8b-instruct-turbo","name":"Meta Llama 3.1 8B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03}},"llama-prompt-guard-2-86m":{"id":"llama-prompt-guard-2-86m","name":"Meta Llama Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Anthropic: Claude 4.5 Haiku (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"sonar":{"id":"sonar","name":"Perplexity Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":1}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"ernie-4.5-21b-a3b-thinking":{"id":"ernie-4.5-21b-a3b-thinking","name":"Baidu Ernie 4.5 21B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-16","last_updated":"2025-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.07,"output":0.28}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Anthropic: Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Google Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998,"cache_write":0.09999999999999999}},"claude-3-haiku-20240307":{"id":"claude-3-haiku-20240307","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-03-07","last_updated":"2024-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Perplexity Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"llama-prompt-guard-2-22m":{"id":"llama-prompt-guard-2-22m","name":"Meta Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"OpenAI GPT-OSS 120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1 Codex Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"o3":{"id":"o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5":{"id":"gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Meta Llama 3.1 8B Instant","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.049999999999999996,"output":0.08}},"sonar-reasoning":{"id":"sonar-reasoning","name":"Perplexity Sonar Reasoning","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":5}},"claude-opus-4":{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}}}},"cortecs":{"id":"cortecs","env":["CORTECS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cortecs.ai/v1","name":"Cortecs","doc":"https://api.cortecs.ai/v1/models","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.898,"output":15.453,"cache_read":0.242}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.996,"output":4.982,"cache_read":0.099,"cache_write":1.186}},"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.179,"output":0.697}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.223,"output":0.39}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.4,"cache_read":0.04}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Gemma 3 is a family of lightweight, multimodal models from Google, supporting text and image inputs, multilingual capabilities, and a 131K context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":110000},"cost":{"input":0.099,"output":0.299}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.625}},"mistral-large-2402":{"id":"mistral-large-2402","name":"mistral-large-2402","description":"Mistral Large (24.02) is Mistral AI’s most advanced language model, built for complex multilingual reasoning, code generation, and deep text understanding.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":4.284,"output":12.952}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":2.659,"output":10.635,"cache_read":1.33}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"qwen3guard-gen-8b","description":"Qwen3Guard-Gen-8B is a large-scale multilingual safety moderation model designed for high-accuracy prompt and response classification.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.35,"cache_read":0.018}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"ministral-8b-2512","description":"Ministral 3 8B is a balanced, efficient multimodal model offering strong text and vision capabilities, optimized for edge and local deployment.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.179,"output":0.179,"cache_read":0.017}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.09,"output":0.17,"cache_read":0.014}},"mistral-7b-instruct-v0.2":{"id":"mistral-7b-instruct-v0.2","name":"mistral-7b-instruct-v0.2","description":"Mistral 7B Instruct is a compact, 7B parameter model optimized for fast and efficient text and code generation with a 32K token context window.","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.159,"output":0.219}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.069,"output":0.455,"cache_read":0.018}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":14.999}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.167,"output":0.891}},"nova-lite-v1":{"id":"nova-lite-v1","name":"nova-lite-v1","description":"Nova Lite is a fast, low-cost multimodal foundation model capable of reasoning over text, images, and video in 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.069,"output":0.275}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Apertus 70B is an open, multilingual language model designed for research, long-context reasoning, and sovereignty-focused AI systems.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":1.393,"output":2.228}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.279,"output":2.192,"cache_read":0.056}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.988,"output":3.164,"cache_read":0.247}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.5,"output":1.499,"cache_read":0.13}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.111,"output":0.434,"cache_read":0.056}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.296,"output":1.186,"cache_read":0.075}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.179,"output":0.697}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"mistral-medium-3.5","description":"Mistral Medium 3.5 is a frontier multimodal 128B model combining reasoning, coding, and instruction-following with strong agentic performance and efficient deployment.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1.532,"output":7.843,"cache_read":0.154}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.06,"output":0.439,"cache_read":0.019}},"claude-4-6-sonnet":{"id":"claude-4-6-sonnet","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.196,"output":15.94,"cache_read":0.32,"cache_write":3.999}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.475,"output":2.97,"cache_read":0.1}},"claude-4-5-sonnet":{"id":"claude-4-5-sonnet","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.989,"output":14.945,"cache_read":0.326,"cache_write":4.078}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.478,"output":2.392,"cache_read":0.045}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"mistral-7b-instruct-v0.3","description":"Mistral-7B-Instruct-v0.3 model is a fine-tuned version of the Mistral 7B base model, optimized for instruction-following tasks. Released in 2023, it is intended for demonstration purposes and does not include built-in guardrails or moderation features.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":127000},"cost":{"input":0.111,"output":0.111}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.067,"output":0.245,"cache_read":0.014}},"mixtral-8x7B-instruct-v0.1":{"id":"mixtral-8x7B-instruct-v0.1","name":"Mixtral 8x7B Instruct v0.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.488,"output":0.758}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":81920},"cost":{"input":0.111,"output":0.557}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.749,"cache_read":0.033}},"voxtral-small-2507":{"id":"voxtral-small-2507","name":"voxtral-small-2507","description":"Voxtral Small is a multimodal model with audio input, combining advanced speech capabilities with strong text performance for transcription, translation, and audio understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.123,"output":0.368,"cache_read":0.012}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":3.46,"cache_read":0.124}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":2.768,"cache_read":0.124}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.498,"cache_read":0.55,"cache_write":6.874}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.724,"output":0.724}},"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"qwen3guard-gen-0.6b","description":"Qwen3Guard-Gen-0.6B is a lightweight multilingual safety moderation model that classifies prompts and responses into safe, controversial, or unsafe categories.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.159,"output":0.638,"cache_read":0.081}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":32768},"cost":{"input":0.167,"output":0.557}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"pixtral-12b-2409","description":"Pixtral 2409 12B is a state-of-the-art multimodal model with 12B parameters and a 400M vision encoder, natively trained on interleaved text and image data. It excels in tasks spanning vision-language reasoning, instruction following, and pure text understanding, making it highly effective for real-world multimodal applications.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-11-09","last_updated":"2024-11-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.223,"output":0.223}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.045,"output":0.167}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.668,"output":2.674}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.649,"output":9.899,"cache_read":0.165,"cache_write":1}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.495,"output":9.964,"cache_read":0.242,"cache_write":0.434}},"hermes-4-405b":{"id":"hermes-4-405b","name":"hermes-4-405b","description":"Hermes 4 405B is a frontier hybrid-mode reasoning model built on Llama 3.1, optimized for advanced logic, math, coding, and structured output generation.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.996,"output":2.989}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.299,"output":2.491,"cache_read":0.029,"cache_write":0.097}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"qwen3-vl-235b-a22b","description":"Qwen3 VL 235B A22B is a 235B-parameter MoE vision-language flagship model (≈22B active) designed for frontier-level multimodal understanding across text, images, documents, and long videos.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.617,"output":3.119,"cache_read":0.052}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":2,"output":3.999,"cache_read":0.5}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":203000},"cost":{"input":0.08,"output":0.478}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.111,"output":0.167}},"claude-opus4-6":{"id":"claude-opus4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.313,"output":26.561,"cache_read":0.531,"cache_write":6.645}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65000},"cost":{"input":2.898,"output":14.493,"cache_read":0.29,"cache_write":3.624}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.395,"output":1.977,"cache_read":0.099}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.219,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.201,"output":0.5,"cache_read":0.05}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.156,"output":0.625,"cache_read":0.016}},"nemotron-nano-v2-12b":{"id":"nemotron-nano-v2-12b","name":"nemotron-nano-v2-12b","description":"NVIDIA Nemotron Nano v2 12B is a 12-billion-parameter multimodal reasoning model designed for advanced video understanding, document intelligence, and visual reasoning, built with a hybrid Transformer-Mamba architecture for high efficiency and low latency.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.24,"output":0.707}},"mistral-small-2503":{"id":"mistral-small-2503","name":"mistral-small-2503","description":"Combines advanced text and vision capabilities with 24 billion parameters, supporting multilingual tasks and long contexts up to 131k tokens, making it versatile for various applications without sacrificing performance.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.111,"output":0.334}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.192,"output":8.769,"cache_read":0.546}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.219,"cache_write":2.749}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.668,"output":4.01}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1000000},"cost":{"input":0.9,"output":3.24,"cache_read":0.189}},"pixtral-large-2502":{"id":"pixtral-large-2502","name":"Pixtral Large (25.02)","description":"Pixtral Large (25.02) is a 124B open-weight multimodal model built on Mistral Large 2, offering advanced image understanding and strong performance across text and code tasks.","family":"pixtral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.993,"output":5.978}},"claude-opus4-7":{"id":"claude-opus4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"minicpm-v-4.5":{"id":"minicpm-v-4.5","name":"minicpm-v-4.5","description":"MiniCPM-V 4.5 is a compact, high-performance vision-language model excelling in video understanding, OCR, and multimodal reasoning with efficient deployment.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.651,"output":1.097}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Optimized for dialogue, this LLM by Meta outperforms other open-source chat models in benchmarks while prioritizing helpfulness and safety.","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.167,"output":0.167}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.384,"output":4.348,"cache_read":0.346}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"qwen3-30b-a3b-instruct-2507","description":"Qwen3-30B-A3B-Instruct-2507 is an advanced Mixture-of-Experts model optimized for reasoning, coding, and multilingual instruction following.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.099,"output":0.299}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"qwen2.5-vl-72b-instruct","description":"Qwen2.5-VL is a powerful vision-language model with advanced capabilities in visual understanding, long video reasoning, and structured output generation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":1.014,"output":1.014}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.613,"output":1.838,"cache_read":0.061}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.434,"output":1.704,"cache_read":0.134}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.082,"cache_write":0.084}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Ministral 3 14B is a frontier-level 14B multimodal model optimized for local deployment, delivering state-of-the-art text and vision reasoning with a 256K context window and strong agentic capabilities.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.24,"output":0.24,"cache_read":0.022}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.12,"output":0.6,"cache_read":0.012,"cache_write":0.15}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.73,"output":3.46,"cache_read":0.432}},"claude-opus-5.5":{"id":"claude-opus-5.5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4.399,"output":21.998,"cache_read":0.219,"cache_write":5.5}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.446,"output":3.008}},"claude-opus4-8":{"id":"claude-opus4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"mistral-small-3.2-24b-instruct-2506","description":"Mistral-Small-3.2-24B-Instruct-2506 is a 24B parameter instruction-tuned model with enhanced long-context support (128k) and state-of-the-art vision understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.1,"output":0.312}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.652,"output":2.57,"cache_read":0.163}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.199,"cache_read":0.219,"cache_write":2.749}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.296,"output":0.495,"cache_read":0.075}},"nova-2-lite":{"id":"nova-2-lite","name":"Nova 2 Lite","description":"Nova 2 Lite is an advanced multimodal reasoning model that combines efficiency and performance, delivering reliable AI for agentic workflows and enterprise applications.","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.373,"output":3.144}},"nova-pro-v1":{"id":"nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.918,"output":3.671}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.089,"output":0.446,"cache_read":0.01}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"ministral-3b-2512","description":"Ministral 3 3B is a compact, efficient multimodal model with strong language, vision capabilities, and ideal for custom fine-tuning.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.123,"output":0.123,"cache_read":0.012}},"claude-opus4-5":{"id":"claude-opus4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.313,"output":26.568,"cache_read":0.531,"cache_write":6.645}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.873,"output":3.306,"cache_read":0.221}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.656,"output":3.3,"cache_read":0.18}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.359,"output":1.435}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.399,"output":21.998,"cache_read":0.44,"cache_write":5.5}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"mistral-nemo-instruct-2407","description":"A 12B parameter, instruct-tuned language model by Mistral AI and NVIDIA, designed for advanced instruction following, multi-turn conversations, and generating text and code across multiple languages.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-07","last_updated":"2024-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.145,"output":0.145,"cache_read":0.014}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":400000,"output":196000},"cost":{"input":0.349,"output":1.405}},"nova-micro-v1":{"id":"nova-micro-v1","name":"nova-micro-v1","description":"Nova Micro is a multilingual text-to-text foundation model with strong reasoning capabilities and broad language coverage across 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.04,"output":0.159}},"codestral-2508":{"id":"codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.368,"output":1.103,"cache_read":0.037}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.4,"output":11.999,"cache_read":0.24,"cache_write":3}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"nvidia-nemotron-3-nano-30b-a3b","description":"Nemotron-Nano-3-30B-A3B is a compact Mixture-of-Experts model optimized for efficient reasoning, chat, and coding, with strong multilingual support and long-context RAG and agent workflows.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-12","last_updated":"2026-01-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.06,"output":0.24}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.272,"output":1.631,"cache_read":0.025,"cache_write":0.082}}}},"agnes":{"id":"agnes","env":["AGNES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apihub.agnes-ai.com/v1","name":"Agnes AI","doc":"https://agnes-ai.com/doc","models":{"agnes-2.5-pro-alpha":{"id":"agnes-2.5-pro-alpha","name":"Agnes 2.5 Pro Alpha","description":"Paid reasoning model for advanced coding, scientific reasoning, long-context analysis, agentic workflows, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.45,"output":0.9,"cache_read":0.0038}},"agnes-2.0-flash":{"id":"agnes-2.0-flash","name":"Agnes 2.0 Flash","description":"Fast and efficient model for agent workflows, tool calling, coding, and image understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-25","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}},"agnes-2.5-flash":{"id":"agnes-2.5-flash","name":"Agnes 2.5 Flash","description":"Upgraded model with improved coding, agent workflows, tool calling, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07","last_updated":"2026-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}}}},"daoxe":{"id":"daoxe","env":["DAOXE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://daoxe.com/v1","name":"DaoXE","doc":"https://daoxe.com/pricing","models":{"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":5}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}}}},"morph":{"id":"morph","env":["MORPH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.morphllm.com/v1","name":"Morph","doc":"https://docs.morphllm.com/api-reference/introduction","models":{"morph-v3-large":{"id":"morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"morph-v3-fast":{"id":"morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"auto":{"id":"auto","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.85,"output":1.55}}}},"openai":{"id":"openai","env":["OPENAI_API_KEY"],"npm":"@ai-sdk/openai","name":"OpenAI","doc":"https://platform.openai.com/docs/models","models":{"chatgpt-image-latest":{"id":"chatgpt-image-latest","name":"chatgpt-image-latest","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"status":"deprecated","cost":{"input":0.5,"output":1.5,"cache_read":0}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-realtime-2.1":{"id":"gpt-realtime-2.1","name":"GPT-Realtime-2.1","description":"Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4,"input_audio":32,"output_audio":64}},"gpt-4o-2024-05-13":{"id":"gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":5,"output":15}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-image-2":{"id":"gpt-image-2","name":"gpt-image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2022-12","release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":30,"output":60}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":100000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-4o-2024-08-06":{"id":"gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"gpt-image-1":{"id":"gpt-image-1","name":"gpt-image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"status":"deprecated"},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-6-luna":{"id":"gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":0.2,"output":1,"cache_read":0.02,"cache_write":0.25},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-image-1-mini":{"id":"gpt-image-1-mini","name":"gpt-image-1-mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"gpt-6-sol":{"id":"gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"o1-pro":{"id":"o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":150,"output":600}}}},"alibaba-coding-plan-cn":{"id":"alibaba-coding-plan-cn","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan (China)","doc":"https://help.aliyun.com/zh/model-studio/coding-plan","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"io-net":{"id":"io-net","env":["IOINTELLIGENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.intelligence.io.solutions/api/v1","name":"IO.NET","doc":"https://io.net/docs/guides/intelligence/io-intelligence","models":{"meta-llama/Llama-3.2-90B-Vision-Instruct":{"id":"meta-llama/Llama-3.2-90B-Vision-Instruct","name":"Llama 3.2 90B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.35,"output":0.4,"cache_read":0.175,"cache_write":0.7}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.13,"output":0.38,"cache_read":0.065,"cache_write":0.26}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":430000,"output":4096},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.3}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen 3 Next 80B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.8,"cache_read":0.05,"cache_write":0.2}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen 3 235B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.11,"output":0.6,"cache_read":0.055,"cache_write":0.22}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen 2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8.75,"cache_read":1,"cache_write":4}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.04,"cache_read":0.01,"cache_write":0.04}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0.25,"cache_write":1}},"mistralai/Devstral-Small-2505":{"id":"mistralai/Devstral-Small-2505","name":"Devstral Small 2505","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"mistralai/Mistral-Large-Instruct-2411":{"id":"mistralai/Mistral-Large-Instruct-2411","name":"Mistral Large Instruct 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":6,"cache_read":1,"cache_write":4}},"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar":{"id":"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":106000,"output":4096},"cost":{"input":0.22,"output":0.95,"cache_read":0.11,"cache_write":0.44}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.55,"output":2.25,"cache_read":0.275,"cache_write":1.1}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-09-05","last_updated":"2024-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.39,"output":1.9,"cache_read":0.195,"cache_write":0.78}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-15","last_updated":"2024-11-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.4,"output":1.75,"cache_read":0.2,"cache_write":0.8}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":4096},"cost":{"input":0.03,"output":0.14,"cache_read":0.015,"cache_write":0.06}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.04,"output":0.4,"cache_read":0.02,"cache_write":0.08}}}},"infomaniak":{"id":"infomaniak","env":["INFOMANIAK_API_KEY","INFOMANIAK_PRODUCT_ID"],"npm":"@ai-sdk/openai-compatible","api":"https://api.infomaniak.com/2/ai/${INFOMANIAK_PRODUCT_ID}/openai/v1","name":"Infomaniak","doc":"https://www.infomaniak.com/en/hosting/ai-services/open-source-models","models":{"mini_lm_l12_v2":{"id":"mini_lm_l12_v2","name":"All-MiniLM-L12-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128,"input":128,"output":384},"cost":{"input":0,"output":0}},"bge_multilingual_gemma2":{"id":"bge_multilingual_gemma2","name":"BGE Multilingual Gemma2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-25","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"input":8000,"output":3584},"cost":{"input":0.08,"output":0}},"swiss-ai/Apertus-v1.5-70B":{"id":"swiss-ai/Apertus-v1.5-70B","name":"Apertus v1.5 70B","description":"Open, ethically-sourced Swiss AI model for multilingual, multimodal chat and instruction following","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-08-01","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":8192},"status":"beta","cost":{"input":0.87,"output":3.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":32768},"cost":{"input":0.25,"output":0.5}},"Qwen/Qwen3.5-122B-A10B-FP8":{"id":"Qwen/Qwen3.5-122B-A10B-FP8","name":"Qwen3.5 122B-A10B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"cost":{"input":0.5,"output":3.97}},"Qwen/Qwen3.5-397B-A17B-FP8":{"id":"Qwen/Qwen3.5-397B-A17B-FP8","name":"Qwen3.5 397B-A17B FP8","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"status":"beta","cost":{"input":0.99,"output":4.46}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.25,"output":0.93}},"mistralai/Ministral-3-14B-Instruct-2512":{"id":"mistralai/Ministral-3-14B-Instruct-2512","name":"Ministral 3 14B Instruct","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":25600},"status":"beta","cost":{"input":0.37,"output":0.5}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"status":"beta","cost":{"input":0.74,"output":3.72}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8","name":"Nemotron 3 Nano 30B A3B FP8","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":262144},"status":"beta","cost":{"input":0.06,"output":0.25}}}},"llmtech":{"id":"llmtech","env":["LLMTECH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmtech.eu/v1","name":"LLM Tech","doc":"https://llmtech.eu/models/qwen3.8-27b","models":{"nvidia/Qwen3.8-27B-NVFP4":{"id":"nvidia/Qwen3.8-27B-NVFP4","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2.09,"cache_read":0.04}}}},"crossmodel":{"id":"crossmodel","env":["CROSSMODEL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.crossmodel.ai/v1","name":"CrossModel","doc":"https://www.crossmodel.ai/docs","models":{"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-5-5":{"id":"anthropic/claude-opus-5-5","name":"Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.215,"output":3.645,"cache_read":0.0405,"cache_write":1.215}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.16,"output":0.64,"cache_read":0.04,"cache_write":0.16}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0.96,"output":2.88,"cache_read":0.048,"cache_write":0.96}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0.15}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.16,"cache_write":0.6,"tiers":[{"input":0.8,"output":3.4,"cache_read":0.2,"cache_write":0.8,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.47,"output":2.16,"cache_read":0.1,"cache_write":0.47,"tiers":[{"input":0.62,"output":2.47,"cache_read":0.13,"cache_write":0.62,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1,"output":3.8,"cache_read":0.2,"cache_write":1,"tiers":[{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":3.7,"cache_read":0.18,"cache_write":0.9,"tiers":[{"input":1.1,"output":4.3,"cache_read":0.27,"cache_write":1.1,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}},"x-ai/grok-4.7":{"id":"x-ai/grok-4.7","name":"Grok 4.7","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-09-21","last_updated":"2026-09-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.6,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6,"cache_write":4}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"cache_write":1,"tiers":[{"input":2,"output":4,"cache_read":0.4,"cache_write":2,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4,"cache_write":2}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"xiaomi/mimo-v2.6-pro":{"id":"xiaomi/mimo-v2.6-pro","name":"MiMo-V2.6-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.47,"output":0.94,"cache_read":0.005,"cache_write":0.47}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.16,"output":0.32,"cache_read":0.004,"cache_write":0.16}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.47,"output":0.94,"cache_read":0.005,"cache_write":0.47}},"xiaomi/mimo-v2.6-flash":{"id":"xiaomi/mimo-v2.6-flash","name":"MiMo-V2.6-Flash","description":"MiMo Flash model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16,"output":0.32,"cache_read":0.004,"cache_write":0.16}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.42}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.33,"tiers":[{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66}}},"gemini/gemini-3.6-flash":{"id":"gemini/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3.5-flash-lite":{"id":"gemini/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gemini/gemini-3.1-pro-preview":{"id":"gemini/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":4}}},"gemini/gemini-3.5-flash":{"id":"gemini/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.5}},"gemini/gemini-2.5-pro":{"id":"gemini/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}}},"gemini/gemini-2.5-flash":{"id":"gemini/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gemini/gemini-3.7-flash":{"id":"gemini/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3-flash-preview":{"id":"gemini/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.5}},"gemini/gemini-3.8-flash":{"id":"gemini/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-2.5-flash-lite":{"id":"gemini/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.1}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.88,"output":5.63,"cache_read":0.375,"cache_write":2.35}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.88,"output":5.63,"cache_read":0.23,"cache_write":2.35}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.04,"output":0.13,"cache_read":0.01,"cache_write":0.04,"tiers":[{"input":0.1,"output":0.37,"cache_read":0.02,"cache_write":0.12,"tier":{"type":"context","size":32000}},{"input":0.19,"output":0.74,"cache_read":0.04,"cache_write":0.24,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3.8-omni-flash":{"id":"qwen/qwen3.8-omni-flash","name":"Qwen3.8 Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"release_date":"2026-09-17","last_updated":"2026-09-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.13,"output":0.43,"cache_read":0.016,"cache_write":0.13}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.19,"output":1.13,"cache_read":0.019,"cache_write":0.24,"tiers":[{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94}}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.13,"output":0.43,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.32,"output":1.88,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.25,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5}}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":10}}},"openai/gpt-6-luna":{"id":"openai/gpt-6-luna","name":"GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"cache_write":0.125,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02,"cache_write":0.25}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-6-sol":{"id":"openai/gpt-6-sol","name":"GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}}}},"arcee":{"id":"arcee","env":["ARCEE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.arcee.ai/api/v1","name":"Arcee","doc":"https://docs.arcee.ai","models":{"trinity-large-thinking":{"id":"trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"status":"beta","cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":3,"output":15,"cache_read":0.3}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"drun":{"id":"drun","env":["DRUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://chat.d.run/v1","name":"D.Run (China)","doc":"https://www.d.run","models":{"public/deepseek-v3":{"id":"public/deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.28,"output":1.1}},"public/minimax-m25":{"id":"public/minimax-m25","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"temperature":true,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.29,"output":1.16}},"public/deepseek-r1":{"id":"public/deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.55,"output":2.2}}}},"amd":{"id":"amd","env":["AMD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://developer.amd.com.cn/radeon/api/v1","name":"AMD","doc":"https://developer.amd.com.cn/radeon/tokenfactory","models":{"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"DeepSeek-V4.1-Flash":{"id":"DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"Qwen3.8-Flash-Next":{"id":"Qwen3.8-Flash-Next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"MiniCPM5-2B":{"id":"MiniCPM5-2B","name":"MiniCPM5-2B","description":"Dense 2B-class open-source model for on-device and resource-constrained use, with native long-context support, tool calling, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-09-06","last_updated":"2026-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.124,"output":0.7425,"cache_read":0.124}},"DeepSeek-V4-Flash-Vision-Exp":{"id":"DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}}}},"qvac":{"id":"qvac","env":["QVAC_API_KEY"],"npm":"@qvac/ai-sdk-provider","name":"QVAC","doc":"https://www.npmjs.com/package/@qvac/ai-sdk-provider","models":{"qwen3.5-0.8b":{"id":"qwen3.5-0.8b","name":"Qwen3.5 0.8B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"Qwen3.5 2B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.5-4b":{"id":"qwen3.5-4b","name":"Qwen3.5 4B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"claudinio":{"id":"claudinio","env":["CLAUDINIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.claudin.io/v1","name":"Claudinio","doc":"https://claudin.io","models":{"claudinio":{"id":"claudinio","name":"Claudinio","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-06-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.5,"output":2,"cache_read":0.15}},"claudius":{"id":"claudius","name":"Claudius","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":8,"cache_read":0.9}}}},"runinfra":{"id":"runinfra","env":["RUNINFRA_GATEWAY_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.runinfra.ai/v1","name":"RunInfra","doc":"https://runinfra.ai/docs","models":{"Inferact/Qwen3.8-2.4T-A95B-NVFP4":{"id":"Inferact/Qwen3.8-2.4T-A95B-NVFP4","name":"Qwen3.8 2.4T A95B (NVFP4)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.2}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.13,"output":0.27,"cache_read":0.01}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.6,"output":1.9,"cache_read":0.03}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.01}},"ornith-ai/Ornith-1.5-35B-A3B":{"id":"ornith-ai/Ornith-1.5-35B-A3B","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}}}},"hetzner":{"id":"hetzner","env":["HETZNER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.hetzner.com/api/v1","name":"Hetzner","doc":"https://experiments.hetzner.com/docs/inference","models":{"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}}}},"digitalocean":{"id":"digitalocean","env":["DIGITALOCEAN_ACCESS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.do-ai.run/v1","name":"DigitalOcean","doc":"https://docs.digitalocean.com/products/gradient-ai-platform/details/models/","models":{"deepseek-3.2":{"id":"deepseek-3.2","name":"Deepseek 3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.5,"output":1.6,"cache_read":0.15}},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"OpenAI GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai-gpt-5.2-pro":{"id":"openai-gpt-5.2-pro","name":"OpenAI GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":21,"output":168}},"all-mini-lm-l6-v2":{"id":"all-mini-lm-l6-v2","name":"All-MiniLM-L6-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256,"output":384},"cost":{"input":0.009,"output":0}},"multi-qa-mpnet-base-dot-v1":{"id":"multi-qa-mpnet-base-dot-v1","name":"Multi-QA-mpnet-base-dot-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":768},"cost":{"input":0.009,"output":0}},"openai-gpt-image-1":{"id":"openai-gpt-image-1","name":"GPT Image 1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"anthropic-claude-3-opus":{"id":"anthropic-claude-3-opus","name":"Claude 3 Opus","description":"Legacy model retained for compatibility with older integrations","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic-claude-4.6-sonnet":{"id":"anthropic-claude-4.6-sonnet","name":"Anthropic Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic-claude-haiku-4.5":{"id":"anthropic-claude-haiku-4.5","name":"Anthropic Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"openai-o1":{"id":"openai-o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"anthropic-claude-opus-4.7":{"id":"anthropic-claude-opus-4.7","name":"Anthropic Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"stable-diffusion-3.5-large":{"id":"stable-diffusion-3.5-large","name":"Stable Diffusion 3.5 Large","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-10-22","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":256,"output":1},"cost":{"input":0.08,"output":0}},"openai-gpt-6-luna":{"id":"openai-gpt-6-luna","name":"OpenAI GPT-6 Luna","description":"OpenAI's most efficient model for focused, high-volume tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05-18","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1,"output":0.5,"cache_read":0.01,"tiers":[{"input":0.2,"output":0.75,"cache_read":0.02,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.2,"output":0.75,"cache_read":0.02}}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8-Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic-claude-opus-5.5":{"id":"anthropic-claude-opus-5.5","name":"Anthropic Claude Opus 5.5","description":"Claude model for long-running agentic coding and knowledge work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.2,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.4,"cache_write":10,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.4,"cache_write":10}}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.25,"output":0.87}},"openai-o3-mini":{"id":"openai-o3-mini","name":"OpenAI o3 mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"glm-5":{"id":"glm-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax M2.5 (Public Preview)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-12","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"anthropic-claude-4.1-opus":{"id":"anthropic-claude-4.1-opus","name":"Anthropic Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"wan2-2-t2v-a14b":{"id":"wan2-2-t2v-a14b","name":"Wan2.2-T2V-A14B","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["video"]},"open_weights":true,"limit":{"context":100,"output":1},"cost":{"input":0.6,"output":0}},"openai-gpt-4o":{"id":"openai-gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"anthropic-claude-4.5-haiku":{"id":"anthropic-claude-4.5-haiku","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":1,"cache_write":1.25}},"anthropic-claude-fable-5.1":{"id":"anthropic-claude-fable-5.1","name":"Anthropic Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32678,"output":8192},"cost":{"input":0.99,"output":0.99}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen3-embedding-0.6b":{"id":"qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":1024},"status":"beta","cost":{"input":0.04,"output":0}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-05-22","last_updated":"2024-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768}},"qwen-2.5-14b-instruct":{"id":"qwen-2.5-14b-instruct","name":"Qwen 2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}},"openai-o3":{"id":"openai-o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai-gpt-5":{"id":"openai-gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai-gpt-5.1-codex-max":{"id":"openai-gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"anthropic-claude-3.5-haiku":{"id":"anthropic-claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-11-05","last_updated":"2024-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"anthropic-claude-3.5-sonnet":{"id":"anthropic-claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-06-20","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.7,"cache_read":0.203}},"anthropic-claude-opus-4":{"id":"anthropic-claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"OpenAI GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"openai-gpt-image-2":{"id":"openai-gpt-image-2","name":"OpenAI GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":8,"output":30}},"openai-gpt-image-1.5":{"id":"openai-gpt-image-1.5","name":"OpenAI GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":5,"output":10,"cache_read":1}},"openai-gpt-4o-mini":{"id":"openai-gpt-4o-mini","name":"OpenAI GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":1.7,"cache_read":0.09}},"deepseek-4-flash":{"id":"deepseek-4-flash","name":"Deepseek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-27","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.9,"output":1.7}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"OpenAI GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"nvidia-nemotron-3-super-120b":{"id":"nvidia-nemotron-3-super-120b","name":"NVIDIA Nemotron 3 Super 120B (Public Preview)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.3,"output":0.65,"cache_read":0.06}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3,"cache_read":0.16}},"mistral-3-14B":{"id":"mistral-3-14B","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.2,"output":0.2}},"anthropic-claude-opus-5":{"id":"anthropic-claude-opus-5","name":"Anthropic Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":0.9}},"nemotron-3-nano-30b":{"id":"nemotron-3-nano-30b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"arcee-trinity-large-thinking":{"id":"arcee-trinity-large-thinking","name":"Arcee Trinity Large Thinking (Public Preview)","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.25,"output":0.9,"cache_read":0.06}},"anthropic-claude-5-sonnet":{"id":"anthropic-claude-5-sonnet","name":"Anthropic Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai-gpt-5.4-pro":{"id":"openai-gpt-5.4-pro","name":"OpenAI GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"llama3-8b-instruct":{"id":"llama3-8b-instruct","name":"Llama 3.1 Instruct (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.198,"output":0.198}},"llama3.3-70b-instruct":{"id":"llama3.3-70b-instruct","name":"Llama 3.3 Instruct (70B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.65,"output":0.65}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.1,"output":0.7,"cache_read":0.02}},"alibaba-qwen3-32b":{"id":"alibaba-qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.55}},"openai-gpt-5.3-codex":{"id":"openai-gpt-5.3-codex","name":"OpenAI GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen 3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":3.5,"cache_read":0.111}},"anthropic-claude-3.7-sonnet":{"id":"anthropic-claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic-claude-opus-4.8":{"id":"anthropic-claude-opus-4.8","name":"Anthropic Claude Opus 4.8","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"OpenAI GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"qwen3-tts-voicedesign":{"id":"qwen3-tts-voicedesign","name":"Qwen3 TTS VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":32768,"output":1}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.21}},"openai-gpt-6-sol":{"id":"openai-gpt-6-sol","name":"OpenAI GPT-6 Sol","description":"OpenAI model for complex coding and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-20","release_date":"2026-09-22","last_updated":"2026-09-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"tiers":[{"input":4,"output":15,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4}}},"anthropic-claude-4.5-sonnet":{"id":"anthropic-claude-4.5-sonnet","name":"Anthropic Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"bge-m3":{"id":"bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.02,"output":0}},"ministral-3-8b-instruct-2512":{"id":"ministral-3-8b-instruct-2512","name":"Ministral 3 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"OpenAI GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"tiers":[{"input":8,"output":30,"cache_read":0.8,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8}}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"openai-gpt-5.4-nano":{"id":"openai-gpt-5.4-nano","name":"OpenAI GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"bge-reranker-v2-m3":{"id":"bge-reranker-v2-m3","name":"BGE Reranker v2 M3","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-12","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1},"cost":{"input":0.01,"output":0}},"nemotron-nano-12b-v2-vl":{"id":"nemotron-nano-12b-v2-vl","name":"Nemotron-nano 12b v2-vl","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.6}},"openai-gpt-5.4-mini":{"id":"openai-gpt-5.4-mini","name":"OpenAI GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"anthropic-claude-opus-4.6":{"id":"anthropic-claude-opus-4.6","name":"Anthropic Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"e5-large-v2":{"id":"e5-large-v2","name":"E5 Large v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-05-19","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.02,"output":0}},"anthropic-claude-opus-4.5":{"id":"anthropic-claude-opus-4.5","name":"Anthropic Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"Deepseek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.348}},"gemma-4-31B-it":{"id":"gemma-4-31B-it","name":"Gemma 4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.18,"output":0.5,"cache_read":0.036}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"OpenAI GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-5.3":{"id":"glm-5.3","name":"GLM5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gte-large-en-v1.5":{"id":"gte-large-en-v1.5","name":"GTE Large (v1.5)","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-27","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.09,"output":0}},"anthropic-claude-fable-5":{"id":"anthropic-claude-fable-5","name":"Anthropic Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai-gpt-oss-20b":{"id":"openai-gpt-oss-20b","name":"OpenAI GPT-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.45}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"OpenAI GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral Nemo Instruct","description":"Legacy model retained for compatibility with older integrations","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.3,"output":0.3}},"anthropic-claude-sonnet-4":{"id":"anthropic-claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75}}},"fal-ai/fast-sdxl":{"id":"fal-ai/fast-sdxl","name":"Fast SDXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-07-26","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}},"fal-ai/elevenlabs/tts/multilingual-v2":{"id":"fal-ai/elevenlabs/tts/multilingual-v2","name":"ElevenLabs Multilingual TTS v2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-08-22","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fal-ai/flux/schnell":{"id":"fal-ai/flux/schnell","name":"FLUX.1 [schnell]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-01","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}},"fal-ai/stable-audio-25/text-to-audio":{"id":"fal-ai/stable-audio-25/text-to-audio","name":"Stable Audio 2.5 (Text-to-Audio)","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-08","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}}}},"aixy":{"id":"aixy","env":["AIXY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aixy-gateway.com/v1","name":"Aixy","doc":"https://docs.aixy-gateway.com/integrations/overview","models":{"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}}}},"alibaba-cn":{"id":"alibaba-cn","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope.aliyuncs.com/compatible-mode/v1","name":"Alibaba (China)","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.216}},"qwen3.5-flash":{"id":"qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-23","last_updated":"2026-09-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.029,"output":0.287,"reasoning":0.287,"tiers":[{"input":0.115,"output":1.147,"reasoning":1.147,"tier":{"type":"context","size":128000}},{"input":0.172,"output":1.72,"reasoning":1.72,"tier":{"type":"context","size":256000}}]}},"qwen2-5-coder-32b-instruct":{"id":"qwen2-5-coder-32b-instruct","name":"Qwen2.5-Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"deepseek-v3-2-exp":{"id":"deepseek-v3-2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.287,"output":0.431}},"deepseek-r1-distill-qwen-14b":{"id":"deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.144,"output":0.431}},"qwen-math-plus":{"id":"qwen-math-plus","name":"Qwen Math Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-08-16","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen-deep-research":{"id":"qwen-deep-research","name":"Qwen Deep Research","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":7.742,"output":23.367}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwq-32b":{"id":"qwq-32b","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.294,"output":6.881}},"deepseek-v3-1":{"id":"deepseek-v3-1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.574,"output":1.721}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.143353,"output":1.433525,"reasoning":4.300576}},"qwen-plus-character":{"id":"qwen-plus-character","name":"Qwen Plus Character","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.115,"output":0.287}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.345,"output":1.377}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":1.434}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.77744,"output":5.33231,"cache_read":0.22218,"cache_write":2.22179}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":2.827,"output":14.133,"cache_read":0.283}},"qwen2-5-math-7b-instruct":{"id":"qwen2-5-math-7b-instruct","name":"Qwen2.5-Math 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.144,"output":0.287}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-09-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.115,"output":0.688,"reasoning":0.688,"tiers":[{"input":0.287,"output":1.72,"reasoning":1.72,"tier":{"type":"context","size":128000}},{"input":0.573,"output":3.44,"reasoning":3.44,"tier":{"type":"context","size":256000}}]}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":1.721}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58,"tiers":[{"input":0.86,"output":3.154,"tier":{"type":"context","size":32000}}]}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.29754,"output":1.19015,"cache_read":0.01488}},"deepseek-r1-distill-llama-8b":{"id":"deepseek-r1-distill-llama-8b","name":"DeepSeek R1 Distill Llama 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwen2-5-coder-7b-instruct":{"id":"qwen2-5-coder-7b-instruct","name":"Qwen2.5-Coder 7B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.287}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.115,"output":0.287,"reasoning":1.147,"cache_read":0.012,"cache_write":0.144}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.02962,"output":0.1185,"cache_read":0.002962,"cache_write":0.03703,"tiers":[{"input":0.08887,"output":0.35549,"cache_read":0.008887,"cache_write":0.11109,"tier":{"type":"context","size":32000}},{"input":0.17774,"output":0.71098,"cache_read":0.017774,"cache_write":0.22218,"tier":{"type":"context","size":256000}}]}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Moonshot Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.929,"output":3.858}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434,"reasoning":1.434,"tiers":[{"input":0.574,"output":2.294,"reasoning":2.294,"tier":{"type":"context","size":32000}},{"input":1.004,"output":4.014,"reasoning":4.014,"tier":{"type":"context","size":128000}}]}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.287,"output":1.147}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.861,"output":3.441,"tiers":[{"input":1.291,"output":5.161,"tier":{"type":"context","size":32000}},{"input":2.151,"output":8.602,"tier":{"type":"context","size":128000}}]}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.044,"output":0.087,"reasoning":0.431}},"deepseek-r1-distill-qwen-7b":{"id":"deepseek-r1-distill-qwen-7b","name":"DeepSeek R1 Distill Qwen 7B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.072,"output":0.144}},"qwen2-5-math-72b-instruct":{"id":"qwen2-5-math-72b-instruct","name":"Qwen2.5-Math 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.216,"output":0.861,"tiers":[{"input":0.323,"output":1.291,"tier":{"type":"context","size":32000}},{"input":0.538,"output":2.151,"tier":{"type":"context","size":128000}}]}},"qwen-doc-turbo":{"id":"qwen-doc-turbo","name":"Qwen Doc Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.087,"output":0.144}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Moonshot Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.574,"output":2.294}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.147,"output":4.588}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.144}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":0.431,"reasoning":1.076}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Moonshot Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.574,"output":2.411}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.115,"output":0.287}},"deepseek-r1-distill-qwen-1-5b":{"id":"deepseek-r1-distill-qwen-1-5b","name":"DeepSeek R1 Distill Qwen 1.5B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.144,"output":0.574}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.087,"output":0.345,"input_audio":5.448}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.259,"output":0.775}},"tongyi-intent-detect-v3":{"id":"tongyi-intent-detect-v3","name":"Tongyi Intent Detect V3","description":"General-purpose chat model for instruction following, writing, and analysis","family":"yi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1024},"cost":{"input":0.058,"output":0.144}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.286705,"output":1.14682,"reasoning":2.867051}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2026-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0.574,"output":2.296,"tiers":[{"input":0.861,"output":3.444,"tier":{"type":"context","size":32000}},{"input":1.435,"output":5.74,"tier":{"type":"context","size":128000}},{"input":2.87,"output":28.7,"tier":{"type":"context","size":256000}}]}},"moonshot-kimi-k2-instruct":{"id":"moonshot-kimi-k2-instruct","name":"Moonshot Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":2.294}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.431}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11875,"output":0.40073,"cache_read":0.01187,"cache_write":0.14844}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.717}},"qwen-math-turbo":{"id":"qwen-math-turbo","name":"Qwen Math Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.287,"output":0.861}},"qwen-long":{"id":"qwen-long","name":"Qwen Long","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"output":8192},"cost":{"input":0.072,"output":0.287}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245800,"output":65536},"cost":{"input":1.32,"output":7.9,"cache_read":0.132}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.101,"output":0.28}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032,"reasoning":1.032,"tiers":[{"input":0.43,"output":2.58,"reasoning":2.58,"tier":{"type":"context","size":128000}}]}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.287,"reasoning":0.717}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1,"output":3.851,"cache_read":0.275,"cache_write":0}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.17,"tiers":[{"input":1.1,"output":3.851,"tier":{"type":"context","size":32000}}]}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.032,"output":0.032}},"deepseek-r1-distill-qwen-32b":{"id":"deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.574,"reasoning":1.434}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.1,"output":3.851,"cache_read":0.275,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2026-09-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.043,"output":0.072}},"siliconflow/deepseek-v3-0324":{"id":"siliconflow/deepseek-v3-0324","name":"siliconflow/deepseek-v3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":1}},"siliconflow/deepseek-v3.1-terminus":{"id":"siliconflow/deepseek-v3.1-terminus","name":"siliconflow/deepseek-v3.1-terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":1}},"siliconflow/deepseek-r1-0528":{"id":"siliconflow/deepseek-r1-0528","name":"siliconflow/deepseek-r1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.18}},"siliconflow/deepseek-v3.2":{"id":"siliconflow/deepseek-v3.2","name":"siliconflow/deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.42}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"kimi/kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"MiniMax/MiniMax-M2.7":{"id":"MiniMax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}}}}}
\ No newline at end of file